{
  "datasets": {
    "osworld-energy50-representative": {
      "name": "osworld-energy50-representative",
      "label": "OSWorld · 50 tasks",
      "href": "dataset-osworld-energy50-representative.html",
      "records": [
        {
          "entry_id": 21,
          "entry_name": "GPT-6 Astra · xhigh · svc_121_b4bca3",
          "model": "GPT-6 Astra",
          "effort": "xhigh",
          "effort_rank": 4,
          "model_type": "closed",
          "performance": 0.916190436008581,
          "time_per_task_sec": 126.82108600532,
          "average_time_per_task_sec": 126.82108600532,
          "median_time_per_task_sec": 98.92762316099999,
          "cost_usd": 0.7081329599999999,
          "turns_per_task": null,
          "release_date": "2026-09-03",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_121_b4bca3",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gpt6astra-energy50-xhigh-modal-v1-121.zip",
          "cost_basis": "Astra_standard_short_context_API_equivalent_not_subscription_bill",
          "notes": "Cost is short-context API-equivalent usage, not a subscription payment; full tool/response trace unavailable in CLI export.",
          "quality_flags": "{\"Astra_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":50,\"codex_model_response_count_and_full_tool_calls_unavailable\":50}",
          "selected_task_seed_set_sha256": "500b8dfdf9f06cb8d05728d859a104e613b8a607283f5d91e9cc537ef41319e2",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.7081329599999999,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Astra_standard_short_context_API_equivalent_not_subscription_bill. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "HF COST-pranjal-eval137.json pinned pricing snapshot; input/cache/write/output=10/1/12.5/50 USD per million",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9b541e31ddb6c7ffcfde1f460f5c2d449f625072/COST-pranjal-eval137.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-03",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-6-astra",
            "notes": "OpenAI's September 3, 2026 changelog announces GPT-6 Astra. Proprietary hosted model.",
            "model": "GPT-6 Astra",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_121_b4bca3"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gpt6astra-energy50-xhigh-modal-v1-121.zip"
          ],
          "task_count": 50,
          "steps_per_task": 7.14,
          "tool_calls_per_task": 19.9,
          "model_responses_per_task": null,
          "variant": "Normal I/O",
          "series": "GPT-6 Astra · Normal I/O",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "codex_cli",
            "benchmark_content_hash": "d1c0140e737122942f3108b2d7b73e2c5f894fca048dc22139e78c5772047237",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "0a1e9a0b6132d82a4ebefed299f1bc30ed8de7436d8df4ee793ce90d8719791a",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 123,
          "entry_name": "Gemini 3.8 Flash · low · svc_59_708d1b",
          "model": "Gemini 3.8 Flash",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.916190436008581,
          "time_per_task_sec": 127.12684103442,
          "average_time_per_task_sec": 127.12684103442,
          "median_time_per_task_sec": 69.085467284,
          "cost_usd": 0.10801328849999998,
          "turns_per_task": 18.52,
          "release_date": "2026-09-02",
          "average_output_tokens": 84.45680345572354,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_59_708d1b",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61afda292c109e49aaa87f9dfc35f4e8bc5363a9/jykoh-gemini38flash-osworld-energy50-representative-modal-low-500steps-59.zip",
          "cost_basis": "recorded_template_price_estimate",
          "notes": "New no-preload Energy50 evaluation; preserve its distinct track and measurement contract when comparing earlier runs. Cost is the logged template estimate from final cumulative snapshots, not an invoice.",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "2f19352958c12ff516242a7950d9d0f53c8a96924812e4c57f85f65fcd8cbfd2",
          "run_terminal_events": "[\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 0.10801328849999998,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "recorded_template_price_estimate. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "recorded_template_price_estimate",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61afda292c109e49aaa87f9dfc35f4e8bc5363a9/jykoh-gemini38flash-osworld-energy50-representative-modal-low-500steps-59.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-02",
            "release_source": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/3-8-flash-and-3-8-flash-cyber/",
            "access_source": "https://ai.google.dev/gemini-api/docs/models",
            "notes": "Proprietary Gemini API model.",
            "model": "Gemini 3.8 Flash",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_59_708d1b"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61afda292c109e49aaa87f9dfc35f4e8bc5363a9/jykoh-gemini38flash-osworld-energy50-representative-modal-low-500steps-59.zip"
          ],
          "task_count": 50,
          "steps_per_task": 16.6,
          "tool_calls_per_task": 20.02,
          "model_responses_per_task": 18.52,
          "variant": "",
          "series": "Gemini 3.8 Flash",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "gemini_new",
            "benchmark_content_hash": "fc402fbaf9f4a14426ee72a7af3ca8b8565b48cf92d52ac53ad19f5df0772876",
            "track_name": "api-cpu-agents1-no-preload",
            "measurement_hash": "47f89119bb0e8086be3679fd64d0124328e04c78154957992dbb350d2dafe18c",
            "eval_algorithm": "shared-agent-no-preload@1",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 9,
          "entry_name": "Claude Opus 5 · high · svc_31_878995",
          "model": "Claude Opus 5",
          "effort": "high",
          "effort_rank": 3,
          "model_type": "closed",
          "performance": 0.916190436008581,
          "time_per_task_sec": 135.9910615843,
          "average_time_per_task_sec": 135.9910615843,
          "median_time_per_task_sec": 82.057938039,
          "cost_usd": 0.66105046,
          "turns_per_task": 20.62,
          "release_date": "2026-07-24",
          "average_output_tokens": 181.13967022308438,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_31_878995",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-claude-opus5-high-osworld-energy50-31.zip",
          "cost_basis": "Claude_standard_global_list_price_estimate_from_logged_usage",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "fc3bdcd7f91d8c7d2a210901794d343677ec109c8457969bf8aae2c28201f552",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.66105046,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Claude_standard_global_list_price_estimate_from_logged_usage. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "HF PR16 price snapshot checked 2026-09-13; Opus5 5/6.25/.5/25; Sonnet5 2/2.5/.2/10 USD per million input/5m-write/read/output",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/discussions/16",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-24",
            "release_source": "https://www.anthropic.com/news/claude-opus-5",
            "access_source": "https://www.anthropic.com/news/claude-opus-5",
            "notes": "Proprietary Claude model, available through Anthropic and hosted partners.",
            "model": "Claude Opus 5",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_31_878995"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-claude-opus5-high-osworld-energy50-31.zip"
          ],
          "task_count": 50,
          "steps_per_task": 19.92,
          "tool_calls_per_task": 26.92,
          "model_responses_per_task": 20.62,
          "variant": "",
          "series": "Claude Opus 5",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "claude",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "50"
          }
        },
        {
          "entry_id": 147,
          "entry_name": "GPT-6 Astra · xhigh · svc_157_c2dbaf",
          "model": "GPT-6 Astra",
          "effort": "xhigh",
          "effort_rank": 4,
          "model_type": "closed",
          "performance": 0.916190436008581,
          "time_per_task_sec": 182.72978732688,
          "average_time_per_task_sec": 182.72978732688,
          "median_time_per_task_sec": 109.528088381,
          "cost_usd": 1.36372984,
          "turns_per_task": null,
          "release_date": "2026-09-03",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_157_c2dbaf",
          "source_archive_url": "results-evidence/svc_157_c2dbaf.json",
          "cost_basis": "Astra_standard_short_context_API_equivalent_not_subscription_bill",
          "notes": "Completed local evaluation 157; all 50 tasks. Single-action mode enforces one GUI action followed by a fresh screenshot before another action. Same model, effort and configured limits as the normal-I/O xhigh baseline; owner-approved current runtime and Codex subscription. Cost is the same historical standard API-equivalent estimate from logged usage, not subscription spending. The source link contains derived task metrics and source hashes; the trajectory archive has not been published to Hugging Face. One GUI action per request, with a fresh screenshot required before the next action.",
          "quality_flags": "{\"Astra_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":50,\"codex_model_response_count_and_full_tool_calls_unavailable\":50}",
          "selected_task_seed_set_sha256": "acb5758a717e7fb47a4dd143b26d94bc2c92a49144431f6f3a9f56e7793f74b0",
          "run_terminal_events": "[\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 1.36372984,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Astra_standard_short_context_API_equivalent_not_subscription_bill. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "HF COST-pranjal-eval137.json pinned pricing snapshot; input/cache/write/output=10/1/12.5/50 USD per million",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/447ff4cd77230ee06ee5b0e864873c43ceb0138c/COST-pranjal-eval137.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-03",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-6-astra",
            "notes": "OpenAI's September 3, 2026 changelog announces GPT-6 Astra. Proprietary hosted model.",
            "model": "GPT-6 Astra",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_157_c2dbaf"
          ],
          "source_archive_urls": [
            "results-evidence/svc_157_c2dbaf.json"
          ],
          "task_count": 50,
          "steps_per_task": 16.5,
          "tool_calls_per_task": 16.5,
          "model_responses_per_task": null,
          "variant": "Single action",
          "series": "GPT-6 Astra · Single action",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "codex_cli",
            "benchmark_content_hash": "fc402fbaf9f4a14426ee72a7af3ca8b8565b48cf92d52ac53ad19f5df0772876",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "34761619b59911dddf1748656f82a27a24896a268da28e7192a6407cad0f67b7",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "Codex subscription",
            "io_variant": "normal",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 124,
          "entry_name": "Gemini 3.8 Flash · medium · svc_61_1ec4be",
          "model": "Gemini 3.8 Flash",
          "effort": "medium",
          "effort_rank": 2,
          "model_type": "closed",
          "performance": 0.916190436008581,
          "time_per_task_sec": 250.10731311282,
          "average_time_per_task_sec": 250.10731311282,
          "median_time_per_task_sec": 116.32829057149999,
          "cost_usd": 0.220861218,
          "turns_per_task": 29.1,
          "release_date": "2026-09-02",
          "average_output_tokens": 184.37319587628866,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_61_1ec4be",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61afda292c109e49aaa87f9dfc35f4e8bc5363a9/jykoh-gemini38flash-osworld-energy50-representative-modal-medium-500steps-61.zip",
          "cost_basis": "recorded_template_price_estimate",
          "notes": "New no-preload Energy50 evaluation; preserve its distinct track and measurement contract when comparing earlier runs. Cost is the logged template estimate from final cumulative snapshots, not an invoice.",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "37159d680033e160c9008b7a88f76352643af09e856776cb183b5d062d752546",
          "run_terminal_events": "[\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 0.220861218,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "recorded_template_price_estimate. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "recorded_template_price_estimate",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61afda292c109e49aaa87f9dfc35f4e8bc5363a9/jykoh-gemini38flash-osworld-energy50-representative-modal-medium-500steps-61.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-02",
            "release_source": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/3-8-flash-and-3-8-flash-cyber/",
            "access_source": "https://ai.google.dev/gemini-api/docs/models",
            "notes": "Proprietary Gemini API model.",
            "model": "Gemini 3.8 Flash",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_61_1ec4be"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61afda292c109e49aaa87f9dfc35f4e8bc5363a9/jykoh-gemini38flash-osworld-energy50-representative-modal-medium-500steps-61.zip"
          ],
          "task_count": 50,
          "steps_per_task": 27.16,
          "tool_calls_per_task": 34.7,
          "model_responses_per_task": 29.1,
          "variant": "",
          "series": "Gemini 3.8 Flash",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "gemini_new",
            "benchmark_content_hash": "fc402fbaf9f4a14426ee72a7af3ca8b8565b48cf92d52ac53ad19f5df0772876",
            "track_name": "api-cpu-agents1-no-preload",
            "measurement_hash": "47f89119bb0e8086be3679fd64d0124328e04c78154957992dbb350d2dafe18c",
            "eval_algorithm": "shared-agent-no-preload@1",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 118,
          "entry_name": "GPT-6 Astra · medium · svc_151_181620",
          "model": "GPT-6 Astra",
          "effort": "medium",
          "effort_rank": 2,
          "model_type": "closed",
          "performance": 0.896190436008581,
          "time_per_task_sec": 90.24478268584,
          "average_time_per_task_sec": 90.24478268584,
          "median_time_per_task_sec": 77.9177101705,
          "cost_usd": 0.60010168,
          "turns_per_task": null,
          "release_date": "2026-09-03",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_151_181620",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/749c2fe74b094150530316ad7b31399751e94bbf/pranjal-gpt6astra-energy50-medium-500steps-no-preload-modal-v1-151.zip",
          "cost_basis": "Astra_standard_short_context_API_equivalent_not_subscription_bill",
          "notes": "Added 2026-09-14 at owner request. Separate no-preload track/measurement contract; not a same-track ranking against older preload runs. Cost uses the prior September11 short-context API-equivalent pricing snapshot, not subscription billing. Full CLI model-response/tool trace remains unavailable. Final-task metrics exclude retained non-final attempts.",
          "quality_flags": "{\"Astra_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":50,\"codex_model_response_count_and_full_tool_calls_unavailable\":50}",
          "selected_task_seed_set_sha256": "4dcdf758d100823212c60ac88d1b4faa3095f6e6868cdb952a16194eb05aff0e",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.60010168,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Astra_standard_short_context_API_equivalent_not_subscription_bill. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "HF COST-pranjal-eval137.json pinned pricing snapshot; input/cache/write/output=10/1/12.5/50 USD per million",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/749c2fe74b094150530316ad7b31399751e94bbf/COST-pranjal-eval151.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-03",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-6-astra",
            "notes": "OpenAI's September 3, 2026 changelog announces GPT-6 Astra. Proprietary hosted model.",
            "model": "GPT-6 Astra",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_151_181620"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/749c2fe74b094150530316ad7b31399751e94bbf/pranjal-gpt6astra-energy50-medium-500steps-no-preload-modal-v1-151.zip"
          ],
          "task_count": 50,
          "steps_per_task": 6.3,
          "tool_calls_per_task": 19.26,
          "model_responses_per_task": null,
          "variant": "Normal I/O",
          "series": "GPT-6 Astra · Normal I/O",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "codex_cli",
            "benchmark_content_hash": "d1c0140e737122942f3108b2d7b73e2c5f894fca048dc22139e78c5772047237",
            "track_name": "api-cpu-agents1-no-preload",
            "measurement_hash": "b61a80d6542471507bf5a8cd8e57db8533379f4232a7cbab3eebf468daccf014",
            "eval_algorithm": "shared-agent-no-preload@1",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "normal_io",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 119,
          "entry_name": "GPT-6 Astra · high · svc_152_8d02f4",
          "model": "GPT-6 Astra",
          "effort": "high",
          "effort_rank": 3,
          "model_type": "closed",
          "performance": 0.896190436008581,
          "time_per_task_sec": 100.75932277378,
          "average_time_per_task_sec": 100.75932277378,
          "median_time_per_task_sec": 88.085385653,
          "cost_usd": 0.6363734799999999,
          "turns_per_task": null,
          "release_date": "2026-09-03",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_152_8d02f4",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/749c2fe74b094150530316ad7b31399751e94bbf/pranjal-gpt6astra-energy50-high-500steps-no-preload-modal-v1-152.zip",
          "cost_basis": "Astra_standard_short_context_API_equivalent_not_subscription_bill",
          "notes": "Added 2026-09-14 at owner request. Separate no-preload track/measurement contract; not a same-track ranking against older preload runs. Cost uses the prior September11 short-context API-equivalent pricing snapshot, not subscription billing. Full CLI model-response/tool trace remains unavailable. Final-task metrics exclude retained non-final attempts. One retained infrastructure attempt is represented separately; final-task API-equivalent total=31.818674 USD, extra attempt=0.263074 USD, all recorded sessions=32.081748 USD.",
          "quality_flags": "{\"Astra_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":50,\"codex_model_response_count_and_full_tool_calls_unavailable\":50}",
          "selected_task_seed_set_sha256": "58ddd80d9d4973f9f06ecb40753b7b78b74c1cfba969148840661eb9c37aa6b7",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.6363734799999999,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Astra_standard_short_context_API_equivalent_not_subscription_bill. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "HF COST-pranjal-eval137.json pinned pricing snapshot; input/cache/write/output=10/1/12.5/50 USD per million",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/749c2fe74b094150530316ad7b31399751e94bbf/COST-pranjal-eval152.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-03",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-6-astra",
            "notes": "OpenAI's September 3, 2026 changelog announces GPT-6 Astra. Proprietary hosted model.",
            "model": "GPT-6 Astra",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_152_8d02f4"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/749c2fe74b094150530316ad7b31399751e94bbf/pranjal-gpt6astra-energy50-high-500steps-no-preload-modal-v1-152.zip"
          ],
          "task_count": 50,
          "steps_per_task": 6.52,
          "tool_calls_per_task": 19.76,
          "model_responses_per_task": null,
          "variant": "Normal I/O",
          "series": "GPT-6 Astra · Normal I/O",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "codex_cli",
            "benchmark_content_hash": "d1c0140e737122942f3108b2d7b73e2c5f894fca048dc22139e78c5772047237",
            "track_name": "api-cpu-agents1-no-preload",
            "measurement_hash": "b61a80d6542471507bf5a8cd8e57db8533379f4232a7cbab3eebf468daccf014",
            "eval_algorithm": "shared-agent-no-preload@1",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "normal_io",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 19,
          "entry_name": "GPT-6 Astra · low · svc_127_422895",
          "model": "GPT-6 Astra",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.896190436008581,
          "time_per_task_sec": 105.18058491566,
          "average_time_per_task_sec": 105.18058491566,
          "median_time_per_task_sec": 88.43301021650001,
          "cost_usd": 0.736727,
          "turns_per_task": null,
          "release_date": "2026-09-03",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_127_422895",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gpt6astra-energy50-low-fastio-modal-v1-127.zip",
          "cost_basis": "Astra_standard_short_context_API_equivalent_not_subscription_bill",
          "notes": "Cost is short-context API-equivalent usage, not a subscription payment; full tool/response trace unavailable in CLI export.",
          "quality_flags": "{\"Astra_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":50,\"codex_model_response_count_and_full_tool_calls_unavailable\":50}",
          "selected_task_seed_set_sha256": "d1f6e735cf0951d3fe9be73701d0c4558a91e6177abc42a7a6c55e7783bc12be",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.736727,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Astra_standard_short_context_API_equivalent_not_subscription_bill. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "HF COST-pranjal-eval137.json pinned pricing snapshot; input/cache/write/output=10/1/12.5/50 USD per million",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9b541e31ddb6c7ffcfde1f460f5c2d449f625072/COST-pranjal-eval137.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-03",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-6-astra",
            "notes": "OpenAI's September 3, 2026 changelog announces GPT-6 Astra. Proprietary hosted model.",
            "model": "GPT-6 Astra",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_127_422895"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gpt6astra-energy50-low-fastio-modal-v1-127.zip"
          ],
          "task_count": 50,
          "steps_per_task": 7.52,
          "tool_calls_per_task": 26.14,
          "model_responses_per_task": null,
          "variant": "Fast I/O",
          "series": "GPT-6 Astra · Fast I/O",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "codex_cli",
            "benchmark_content_hash": "d1c0140e737122942f3108b2d7b73e2c5f894fca048dc22139e78c5772047237",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "2a8e7d9d5440fdd01f55b4841b8d25e116a63348c7e3148cdc7f47fa8a630604",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "fast_io",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 11,
          "entry_name": "Claude Sonnet 5 · high · svc_24_ddcb3b",
          "model": "Claude Sonnet 5",
          "effort": "high",
          "effort_rank": 3,
          "model_type": "closed",
          "performance": 0.896190436008581,
          "time_per_task_sec": 140.79131618998,
          "average_time_per_task_sec": 140.79131618998,
          "median_time_per_task_sec": 77.968122628,
          "cost_usd": 0.308575228,
          "turns_per_task": 22.28,
          "release_date": "2026-06-30",
          "average_output_tokens": 217.2423698384201,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_24_ddcb3b",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-claude-sonnet5-high-osworld-energy50-24.zip",
          "cost_basis": "Claude_standard_global_list_price_estimate_from_logged_usage",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "4707deb3863de1ac287e3ee467471cc0b8a0919d19acbeb8efefabfcf3457a24",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.308575228,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Claude_standard_global_list_price_estimate_from_logged_usage. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "HF PR16 price snapshot checked 2026-09-13; Opus5 5/6.25/.5/25; Sonnet5 2/2.5/.2/10 USD per million input/5m-write/read/output",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/discussions/16",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-06-30",
            "release_source": "https://www.anthropic.com/news/claude-sonnet-5",
            "access_source": "https://www.anthropic.com/news/claude-sonnet-5",
            "notes": "Proprietary Claude model, available through Anthropic and hosted partners.",
            "model": "Claude Sonnet 5",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_24_ddcb3b"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-claude-sonnet5-high-osworld-energy50-24.zip"
          ],
          "task_count": 50,
          "steps_per_task": 20.24,
          "tool_calls_per_task": 29.2,
          "model_responses_per_task": 22.28,
          "variant": "",
          "series": "Claude Sonnet 5",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "claude",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "50"
          }
        },
        {
          "entry_id": 27,
          "entry_name": "Gemini 3.7 Flash · medium · svc_85_948d7f",
          "model": "Gemini 3.7 Flash",
          "effort": "medium",
          "effort_rank": 2,
          "model_type": "closed",
          "performance": 0.896190436008581,
          "time_per_task_sec": 198.94348658978,
          "average_time_per_task_sec": 198.94348658978,
          "median_time_per_task_sec": 87.399350148,
          "cost_usd": 0.9436632,
          "turns_per_task": 21.36,
          "release_date": "2026-08-13",
          "average_output_tokens": 168.23876404494382,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_85_948d7f",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gemini37flash-energy50-medium-modal-v2-85.zip",
          "cost_basis": "sum_of_rounded_per_response_template_estimates",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "1b618c33f9b52b16ff231b4ae236dd22d3e283664ced5176d333682063a43030",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.9436632,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "sum_of_rounded_per_response_template_estimates. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "sum_of_rounded_per_response_template_estimates",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gemini37flash-energy50-medium-modal-v2-85.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-08-13",
            "release_source": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/introducing-gemini-3-7-flash/",
            "access_source": "https://ai.google.dev/gemini-api/docs/models",
            "notes": "Proprietary Gemini API model.",
            "model": "Gemini 3.7 Flash",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_85_948d7f"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gemini37flash-energy50-medium-modal-v2-85.zip"
          ],
          "task_count": 50,
          "steps_per_task": 19.44,
          "tool_calls_per_task": 23.34,
          "model_responses_per_task": 21.36,
          "variant": "",
          "series": "Gemini 3.7 Flash",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "gemini35",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 25,
          "entry_name": "Gemini 3.7 Flash · high · svc_86_9e7a56",
          "model": "Gemini 3.7 Flash",
          "effort": "high",
          "effort_rank": 3,
          "model_type": "closed",
          "performance": 0.896190436008581,
          "time_per_task_sec": 254.29542270009998,
          "average_time_per_task_sec": 254.29542270009998,
          "median_time_per_task_sec": 119.742710789,
          "cost_usd": 1.4120642,
          "turns_per_task": 26.2,
          "release_date": "2026-08-13",
          "average_output_tokens": 218.72290076335878,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_86_9e7a56",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gemini37flash-energy50-high-modal-v1-86.zip",
          "cost_basis": "sum_of_rounded_per_response_template_estimates",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "97b703541db04f4fe2241fad6c894374e7261e7ab2c2d090fd74d9b3c5917682",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 1.4120642,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "sum_of_rounded_per_response_template_estimates. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "sum_of_rounded_per_response_template_estimates",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gemini37flash-energy50-high-modal-v1-86.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-08-13",
            "release_source": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/introducing-gemini-3-7-flash/",
            "access_source": "https://ai.google.dev/gemini-api/docs/models",
            "notes": "Proprietary Gemini API model.",
            "model": "Gemini 3.7 Flash",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_86_9e7a56"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gemini37flash-energy50-high-modal-v1-86.zip"
          ],
          "task_count": 50,
          "steps_per_task": 24.26,
          "tool_calls_per_task": 28.3,
          "model_responses_per_task": 26.2,
          "variant": "",
          "series": "Gemini 3.7 Flash",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "gemini35",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 33,
          "entry_name": "Kimi K3 · max · svc_141_f28fed",
          "model": "Kimi K3",
          "effort": "max",
          "effort_rank": 5,
          "model_type": "open",
          "performance": 0.896190436008581,
          "time_per_task_sec": 443.73780842537997,
          "average_time_per_task_sec": 443.73780842537997,
          "median_time_per_task_sec": 216.90054840850001,
          "cost_usd": 0.503421264,
          "turns_per_task": 11.68,
          "release_date": "2026-07-16",
          "average_output_tokens": 893.3681506849316,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_141_f28fed",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9d7e011c785d433d2329ec51785aa3b6ebbee312/pranjal-kimi-k3-parallel-energy50-max-100steps-modal-v1-141.zip",
          "cost_basis": "recorded_provider_usage_cost",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "de8aaadb06f2abad671c609d6051a596ba4f35f8de16a4f2b178715b2a47ec86",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.503421264,
          "cost_kind": "recorded",
          "cost_label": "Recorded usage",
          "cost_note": "recorded_provider_usage_cost. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "recorded_provider_usage_cost",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9d7e011c785d433d2329ec51785aa3b6ebbee312/pranjal-kimi-k3-parallel-energy50-max-100steps-modal-v1-141.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "open",
            "release_date": "2026-07-16",
            "release_source": "https://www.kimi.com/en/blog/kimi-k3",
            "access_source": "https://huggingface.co/moonshotai/Kimi-K3",
            "notes": "Public model launch July 16; downloadable weights followed later. API use in the evaluation does not make this a closed-weight model.",
            "model": "Kimi K3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_141_f28fed"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9d7e011c785d433d2329ec51785aa3b6ebbee312/pranjal-kimi-k3-parallel-energy50-max-100steps-modal-v1-141.zip"
          ],
          "task_count": 50,
          "steps_per_task": 11.7,
          "tool_calls_per_task": 53.58,
          "model_responses_per_task": 11.68,
          "variant": "Batched tool calls",
          "series": "Kimi K3 · Batched tool calls",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "kimi_parallel",
            "benchmark_content_hash": "d1c0140e737122942f3108b2d7b73e2c5f894fca048dc22139e78c5772047237",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "34761619b59911dddf1748656f82a27a24896a268da28e7192a6407cad0f67b7",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 10,
          "entry_name": "Claude Opus 5 · low · svc_32_909005",
          "model": "Claude Opus 5",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.8761904360085809,
          "time_per_task_sec": 85.87818153622,
          "average_time_per_task_sec": 85.87818153622,
          "median_time_per_task_sec": 50.67206232149999,
          "cost_usd": 0.38961248,
          "turns_per_task": 16.08,
          "release_date": "2026-07-24",
          "average_output_tokens": 111.63805970149255,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_32_909005",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-claude-opus5-low-osworld-energy50-32.zip",
          "cost_basis": "Claude_standard_global_list_price_estimate_from_logged_usage",
          "notes": "",
          "quality_flags": "{\"Claude_cost_assumes_standard_global_for_responses_with_missing_tier_geo\":1}",
          "selected_task_seed_set_sha256": "c5d1d2715c7e06e2d17a0b86184d269f49f9ee5aabdb850a680d259f4c2d4f1b",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.38961248,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Claude_standard_global_list_price_estimate_from_logged_usage. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "HF PR16 price snapshot checked 2026-09-13; Opus5 5/6.25/.5/25; Sonnet5 2/2.5/.2/10 USD per million input/5m-write/read/output",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/discussions/16",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-24",
            "release_source": "https://www.anthropic.com/news/claude-opus-5",
            "access_source": "https://www.anthropic.com/news/claude-opus-5",
            "notes": "Proprietary Claude model, available through Anthropic and hosted partners.",
            "model": "Claude Opus 5",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_32_909005"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-claude-opus5-low-osworld-energy50-32.zip"
          ],
          "task_count": 50,
          "steps_per_task": 14.04,
          "tool_calls_per_task": 23.8,
          "model_responses_per_task": 16.08,
          "variant": "",
          "series": "Claude Opus 5",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "claude",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "50"
          }
        },
        {
          "entry_id": 20,
          "entry_name": "GPT-6 Astra · low · svc_122_93c9c5",
          "model": "GPT-6 Astra",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.8761904360085809,
          "time_per_task_sec": 86.11104813758,
          "average_time_per_task_sec": 86.11104813758,
          "median_time_per_task_sec": 68.181259227,
          "cost_usd": 0.56648392,
          "turns_per_task": null,
          "release_date": "2026-09-03",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_122_93c9c5",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gpt6astra-energy50-low-modal-v1-122.zip",
          "cost_basis": "Astra_standard_short_context_API_equivalent_not_subscription_bill",
          "notes": "Cost is short-context API-equivalent usage, not a subscription payment; full tool/response trace unavailable in CLI export.",
          "quality_flags": "{\"Astra_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":50,\"codex_model_response_count_and_full_tool_calls_unavailable\":50}",
          "selected_task_seed_set_sha256": "1e6973339a81c224cd4304af14a76e9332433a8c2eb70b03efd5da1306dceda3",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.56648392,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Astra_standard_short_context_API_equivalent_not_subscription_bill. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "HF COST-pranjal-eval137.json pinned pricing snapshot; input/cache/write/output=10/1/12.5/50 USD per million",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9b541e31ddb6c7ffcfde1f460f5c2d449f625072/COST-pranjal-eval137.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-03",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-6-astra",
            "notes": "OpenAI's September 3, 2026 changelog announces GPT-6 Astra. Proprietary hosted model.",
            "model": "GPT-6 Astra",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_122_93c9c5"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gpt6astra-energy50-low-modal-v1-122.zip"
          ],
          "task_count": 50,
          "steps_per_task": 5.9,
          "tool_calls_per_task": 16.22,
          "model_responses_per_task": null,
          "variant": "Normal I/O",
          "series": "GPT-6 Astra · Normal I/O",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "codex_cli",
            "benchmark_content_hash": "d1c0140e737122942f3108b2d7b73e2c5f894fca048dc22139e78c5772047237",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "0a1e9a0b6132d82a4ebefed299f1bc30ed8de7436d8df4ee793ce90d8719791a",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 26,
          "entry_name": "Gemini 3.7 Flash · low · svc_87_735f9a",
          "model": "Gemini 3.7 Flash",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.8761904360085809,
          "time_per_task_sec": 207.32472104960002,
          "average_time_per_task_sec": 207.32472104960002,
          "median_time_per_task_sec": 84.9894383865,
          "cost_usd": 0.926449,
          "turns_per_task": 20.14,
          "release_date": "2026-08-13",
          "average_output_tokens": 146.1658391261172,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_87_735f9a",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gemini37flash-energy50-low-modal-v1-87.zip",
          "cost_basis": "sum_of_rounded_per_response_template_estimates",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "b1c594bcc9c4b87554be83a2c84cccaf743e20715de0e4eaf210d591b82341de",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.926449,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "sum_of_rounded_per_response_template_estimates. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "sum_of_rounded_per_response_template_estimates",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gemini37flash-energy50-low-modal-v1-87.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-08-13",
            "release_source": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/introducing-gemini-3-7-flash/",
            "access_source": "https://ai.google.dev/gemini-api/docs/models",
            "notes": "Proprietary Gemini API model.",
            "model": "Gemini 3.7 Flash",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_87_735f9a"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gemini37flash-energy50-low-modal-v1-87.zip"
          ],
          "task_count": 50,
          "steps_per_task": 18.24,
          "tool_calls_per_task": 20.88,
          "model_responses_per_task": 20.14,
          "variant": "",
          "series": "Gemini 3.7 Flash",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "gemini35",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 122,
          "entry_name": "Gemini 3.8 Flash · high · svc_60_8db123",
          "model": "Gemini 3.8 Flash",
          "effort": "high",
          "effort_rank": 3,
          "model_type": "closed",
          "performance": 0.8761904360085809,
          "time_per_task_sec": 347.69817593888,
          "average_time_per_task_sec": 347.69817593888,
          "median_time_per_task_sec": 179.9781268635,
          "cost_usd": 0.34745266649999995,
          "turns_per_task": 39.08,
          "release_date": "2026-09-02",
          "average_output_tokens": 224.8930399181167,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_60_8db123",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61afda292c109e49aaa87f9dfc35f4e8bc5363a9/jykoh-gemini38flash-osworld-energy50-representative-modal-high-500steps-60.zip",
          "cost_basis": "recorded_template_price_estimate",
          "notes": "New no-preload Energy50 evaluation; preserve its distinct track and measurement contract when comparing earlier runs. Cost is the logged template estimate from final cumulative snapshots, not an invoice.",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "153f4131e9ea448cfaf1aba3c6025af633a6a887a01cc43f496b068a9b869c2e",
          "run_terminal_events": "[\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 0.34745266649999995,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "recorded_template_price_estimate. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "recorded_template_price_estimate",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61afda292c109e49aaa87f9dfc35f4e8bc5363a9/jykoh-gemini38flash-osworld-energy50-representative-modal-high-500steps-60.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-02",
            "release_source": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/3-8-flash-and-3-8-flash-cyber/",
            "access_source": "https://ai.google.dev/gemini-api/docs/models",
            "notes": "Proprietary Gemini API model.",
            "model": "Gemini 3.8 Flash",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_60_8db123"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61afda292c109e49aaa87f9dfc35f4e8bc5363a9/jykoh-gemini38flash-osworld-energy50-representative-modal-high-500steps-60.zip"
          ],
          "task_count": 50,
          "steps_per_task": 37.18,
          "tool_calls_per_task": 51.46,
          "model_responses_per_task": 39.08,
          "variant": "",
          "series": "Gemini 3.8 Flash",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "gemini_new",
            "benchmark_content_hash": "fc402fbaf9f4a14426ee72a7af3ca8b8565b48cf92d52ac53ad19f5df0772876",
            "track_name": "api-cpu-agents1-no-preload",
            "measurement_hash": "47f89119bb0e8086be3679fd64d0124328e04c78154957992dbb350d2dafe18c",
            "eval_algorithm": "shared-agent-no-preload@1",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 34,
          "entry_name": "Kimi K3 · max · svc_56_9e002a",
          "model": "Kimi K3",
          "effort": "max",
          "effort_rank": 5,
          "model_type": "open",
          "performance": 0.8561904360085809,
          "time_per_task_sec": 606.2705898115,
          "average_time_per_task_sec": 606.2705898115,
          "median_time_per_task_sec": 291.077490049,
          "cost_usd": 0.715786356,
          "turns_per_task": 13.98,
          "release_date": "2026-07-16",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_56_9e002a",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-kimi-k3-unanimous295-11h-56.zip",
          "cost_basis": "recorded_provider_usage_cost",
          "notes": "Published Energy50 max-reasoning subset of Kimi evaluation56; not a separate execution. One model tool call per response; the code in that call can contain multiple GUI actions.",
          "quality_flags": "{\"Kimi_finish_rejected_response_usage_included_but_tool_bodies_unlogged\":1,\"Kimi_noncompletion_replies_have_no_logged_usage_or_body\":2}",
          "selected_task_seed_set_sha256": "a3209ab8908d4673404b00e781c5d0b441f12938ce169a55e367b29314e42ad1",
          "run_terminal_events": "[]",
          "row_type": "published_subset",
          "source_cost_usd": 0.715786356,
          "cost_kind": "recorded",
          "cost_label": "Recorded usage",
          "cost_note": "recorded_provider_usage_cost. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "recorded_provider_usage_cost",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-kimi-k3-unanimous295-11h-56.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "open",
            "release_date": "2026-07-16",
            "release_source": "https://www.kimi.com/en/blog/kimi-k3",
            "access_source": "https://huggingface.co/moonshotai/Kimi-K3",
            "notes": "Public model launch July 16; downloadable weights followed later. API use in the evaluation does not make this a closed-weight model.",
            "model": "Kimi K3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_56_9e002a"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-kimi-k3-unanimous295-11h-56.zip"
          ],
          "task_count": 50,
          "steps_per_task": 12.18,
          "tool_calls_per_task": 61.24,
          "model_responses_per_task": 13.98,
          "variant": "Single tool call",
          "series": "Kimi K3 · Single tool call",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-unanimous-295",
            "benchmark_version": "0.1",
            "template_name": "kimi_k3",
            "benchmark_content_hash": "6f30355e7cc33aba1b9924ca7ea1ab0cf7f422d6dbc30b7e0896cba39edc09b9",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "f411228c5c1556f802fc9938cc7941aee74bfd26531b97e0e73e863c3be4af49",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 15,
          "entry_name": "GPT-5.6 Luna · medium · svc_20_9456b8",
          "model": "GPT-5.6 Luna",
          "effort": "medium",
          "effort_rank": 2,
          "model_type": "closed",
          "performance": 0.8161904360085809,
          "time_per_task_sec": 100.55812353380001,
          "average_time_per_task_sec": 100.55812353380001,
          "median_time_per_task_sec": 53.53789824999999,
          "cost_usd": 0.0255885,
          "turns_per_task": 15.12,
          "release_date": "2026-07-09",
          "average_output_tokens": 214.25925925925927,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_20_9456b8",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gpt56luna-medium-osworld-energy50-20.zip",
          "cost_basis": "sum_of_rounded_per_response_template_estimates",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "59e7b938127e09ca6b1a98e3fc25a98db352b961daa6eefcbfb6a7e21fa02b09",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.0255885,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "sum_of_rounded_per_response_template_estimates. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "sum_of_rounded_per_response_template_estimates",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gpt56luna-medium-osworld-energy50-20.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
            "notes": "OpenAI's July 9, 2026 changelog announces public availability of GPT-5.6 Sol and Luna. The legacy June 26 date is not the public release date. Proprietary hosted model.",
            "model": "GPT-5.6 Luna",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_20_9456b8"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gpt56luna-medium-osworld-energy50-20.zip"
          ],
          "task_count": 50,
          "steps_per_task": 14.16,
          "tool_calls_per_task": 39.78,
          "model_responses_per_task": 15.12,
          "variant": "Direct API",
          "series": "GPT-5.6 Luna · Direct API",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "gpt54",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "50"
          }
        },
        {
          "entry_id": 18,
          "entry_name": "GPT-5.6 Sol · xhigh · svc_23_753ef9",
          "model": "GPT-5.6 Sol",
          "effort": "xhigh",
          "effort_rank": 4,
          "model_type": "closed",
          "performance": 0.8161904360085809,
          "time_per_task_sec": 108.3138664727,
          "average_time_per_task_sec": 108.3138664727,
          "median_time_per_task_sec": 60.93047482199998,
          "cost_usd": 0.52665238,
          "turns_per_task": 13.62,
          "release_date": "2026-07-09",
          "average_output_tokens": 176.3509544787078,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_23_753ef9",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gpt56sol-xhigh-osworld-energy50-23.zip",
          "cost_basis": "sum_of_rounded_per_response_template_estimates",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "121b304fa6b1a89e277ee2158a1662f7ebd8fc086e589810ffdb0d4a99ba1858",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.52665238,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "sum_of_rounded_per_response_template_estimates. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "sum_of_rounded_per_response_template_estimates",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gpt56sol-xhigh-osworld-energy50-23.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-5.6-sol",
            "notes": "OpenAI's July 9, 2026 changelog announces public availability of GPT-5.6 Sol and Luna. The legacy June 26 date is not the public release date. Proprietary hosted model.",
            "model": "GPT-5.6 Sol",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_23_753ef9"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gpt56sol-xhigh-osworld-energy50-23.zip"
          ],
          "task_count": 50,
          "steps_per_task": 12.66,
          "tool_calls_per_task": 36.78,
          "model_responses_per_task": 13.62,
          "variant": "",
          "series": "GPT-5.6 Sol",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "gpt54",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "50"
          }
        },
        {
          "entry_id": 29,
          "entry_name": "Kimi K3 · high · svc_90_357e81",
          "model": "Kimi K3",
          "effort": "high",
          "effort_rank": 3,
          "model_type": "open",
          "performance": 0.8161904360085809,
          "time_per_task_sec": 305.07266010828,
          "average_time_per_task_sec": 305.07266010828,
          "median_time_per_task_sec": 145.162273211,
          "cost_usd": 0.414053082,
          "turns_per_task": 10.7,
          "release_date": "2026-07-16",
          "average_output_tokens": 707.1532710280375,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_90_357e81",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-kimi-k3-energy50-high-modal-v1-90.zip",
          "cost_basis": "recorded_provider_usage_cost",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "25414f8360c7e436829a61d95e3ddc0a5f90099fa9cb993540c736c9b0d6d675",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.414053082,
          "cost_kind": "recorded",
          "cost_label": "Recorded usage",
          "cost_note": "recorded_provider_usage_cost. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "recorded_provider_usage_cost",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-kimi-k3-energy50-high-modal-v1-90.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "open",
            "release_date": "2026-07-16",
            "release_source": "https://www.kimi.com/en/blog/kimi-k3",
            "access_source": "https://huggingface.co/moonshotai/Kimi-K3",
            "notes": "Public model launch July 16; downloadable weights followed later. API use in the evaluation does not make this a closed-weight model.",
            "model": "Kimi K3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_90_357e81"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-kimi-k3-energy50-high-modal-v1-90.zip"
          ],
          "task_count": 50,
          "steps_per_task": 8.92,
          "tool_calls_per_task": 113.6,
          "model_responses_per_task": 10.7,
          "variant": "Single tool call",
          "series": "Kimi K3 · Single tool call",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "kimi_k3",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 17,
          "entry_name": "GPT-5.6 Sol · medium · svc_22_3a244d",
          "model": "GPT-5.6 Sol",
          "effort": "medium",
          "effort_rank": 2,
          "model_type": "closed",
          "performance": 0.796190436008581,
          "time_per_task_sec": 121.34443960266,
          "average_time_per_task_sec": 121.34443960266,
          "median_time_per_task_sec": 68.1446988055,
          "cost_usd": 0.539377,
          "turns_per_task": 13.68,
          "release_date": "2026-07-09",
          "average_output_tokens": 198.5979532163743,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_22_3a244d",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gpt56sol-medium-osworld-energy50-22.zip",
          "cost_basis": "sum_of_rounded_per_response_template_estimates",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "deaaf97ada8116c8e707c2cdd90612f6f0a607fd052609149c1a6d6221964fb6",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.539377,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "sum_of_rounded_per_response_template_estimates. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "sum_of_rounded_per_response_template_estimates",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gpt56sol-medium-osworld-energy50-22.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-5.6-sol",
            "notes": "OpenAI's July 9, 2026 changelog announces public availability of GPT-5.6 Sol and Luna. The legacy June 26 date is not the public release date. Proprietary hosted model.",
            "model": "GPT-5.6 Sol",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_22_3a244d"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gpt56sol-medium-osworld-energy50-22.zip"
          ],
          "task_count": 50,
          "steps_per_task": 12.72,
          "tool_calls_per_task": 42.64,
          "model_responses_per_task": 13.68,
          "variant": "",
          "series": "GPT-5.6 Sol",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "gpt54",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "50"
          }
        },
        {
          "entry_id": 137,
          "entry_name": "Yutori n2 · xhigh · svc_13_01885e",
          "model": "Yutori n2",
          "effort": "xhigh",
          "effort_rank": 4,
          "model_type": "closed",
          "performance": 0.796190436008581,
          "time_per_task_sec": 224.9677141393,
          "average_time_per_task_sec": 224.9677141393,
          "median_time_per_task_sec": 136.968442288,
          "cost_usd": 0.103612176,
          "turns_per_task": 19.72,
          "release_date": "2026-08-26",
          "average_output_tokens": 380.2728194726166,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_13_01885e",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61a43287ed7d5eb5aa625e0daa11157dbb5004e1/pranjal-yutori-n2-energy50-xhigh-500steps-no-preload-modal-v1-13.zip",
          "cost_basis": "published_API_rate_estimate_from_logged_usage_and_billed_cache_not_recorded_charge",
          "notes": "Published HF PR23, merged before inventory. All 50 final tasks complete; original no-preload contract and GUI-only tool configuration retained. Full response usage recovered from agent.stdout despite the published README marking usage unavailable; duplicated SDK step usage is not counted again. Cost is an API-rate estimate using logged input/output and billed cached input at publisher rates verified September 19, not a recorded charge. Reasoning-token breakdown is incomplete and remains unknown.",
          "quality_flags": "{\"reasoning_token_breakdown_not_logged_for_all_responses\":50}",
          "selected_task_seed_set_sha256": "43559840deff0d72c8117891dcec81ca085d7a564751914cc08e2d47b40393d2",
          "run_terminal_events": "[\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 0.103612176,
          "cost_kind": "source_estimate",
          "cost_label": "API estimate · logged usage",
          "cost_note": "published_API_rate_estimate_from_logged_usage_and_billed_cache_not_recorded_charge. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://yutori.com/blog/introducing-n2 verified 2026-09-19; input/cached input/output=0.50/0.05/4.00 USD per million",
          "cost_source_url": "https://yutori.com/blog/introducing-n2",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "verified_on": "2026-09-19",
            "model_type": "closed",
            "release_date": "2026-08-26",
            "release_source": "https://yutori.com/blog/introducing-n2",
            "access_source": "https://docs.yutori.com/reference/n2",
            "notes": "Publisher's August 26 launch announces Navigator n2 through the Yutori API. Archived response.model is n2; the evaluated GUI-only SDK configuration disables bash/read/write/edit. Sources checked September 19, 2026.",
            "model": "Yutori n2"
          },
          "source_run_ids": [
            "svc_13_01885e"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61a43287ed7d5eb5aa625e0daa11157dbb5004e1/pranjal-yutori-n2-energy50-xhigh-500steps-no-preload-modal-v1-13.zip"
          ],
          "task_count": 50,
          "steps_per_task": 17.56,
          "tool_calls_per_task": 48.16,
          "model_responses_per_task": 19.72,
          "variant": "",
          "series": "Yutori n2",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "yutori_n2",
            "benchmark_content_hash": "fc402fbaf9f4a14426ee72a7af3ca8b8565b48cf92d52ac53ad19f5df0772876",
            "track_name": "api-cpu-agents1-no-preload",
            "measurement_hash": "999eacc0929440c65c20fb3785b102d69e756a52098e80f742f769e237bb2996",
            "eval_algorithm": "shared-agent-no-preload@1",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 131,
          "entry_name": "GPT-5.6 Luna · high · svc_155_7fbf62",
          "model": "GPT-5.6 Luna",
          "effort": "high",
          "effort_rank": 3,
          "model_type": "closed",
          "performance": 0.796190436008581,
          "time_per_task_sec": 324.83747465096,
          "average_time_per_task_sec": 324.83747465096,
          "median_time_per_task_sec": 194.34112513450003,
          "cost_usd": 0.075945944,
          "turns_per_task": null,
          "release_date": "2026-07-09",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_155_7fbf62",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/ac8db5465e7efaef62b38376760a9e70c459f99f/pranjal-gpt56luna-codex-energy50-high-500steps-no-preload-modal-v1-155.zip",
          "cost_basis": "Luna_standard_short_context_API_equivalent_not_subscription_bill",
          "notes": "Added 2026-09-16 at owner request. GPT-5.6 Luna via Codex, not the direct-API gpt54 template. All 50 original tasks included, including failures. No rerun or regrading. No-preload track/measurement contract; not a same-track ranking against older preload runs. Costs are standard short-context API-equivalent estimates using official 2026-09-16 Luna rates, not subscription charges. All 50 session token summaries complete; cached input is a subset of input, reasoning is a subset of output, recorded cache-write counters are zero. Full CLI model-response/tool trace remains unavailable; observed CLI tool items and harness steps are separate metrics. No additional retained attempts found. Submitted agent.py SHA-256=306fe8c02f9c85a5ff90be95d8efef897a3555ab37fe4f46e84dad68dad0058a. Full per-task normalized evidence and usage breakdowns are linked in companion_cost_usage_url.",
          "quality_flags": "{\"Luna_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":50,\"codex_model_response_count_and_full_tool_calls_unavailable\":50}",
          "selected_task_seed_set_sha256": "7e437e390b99e4b649be1637a9157c4f8e873e21d6a6d9e81d35f651748318e5",
          "run_terminal_events": "[\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 0.075945944,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Luna_standard_short_context_API_equivalent_not_subscription_bill. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges; assumes standard short-context rates; per-request long-context/fast-mode adjustments and tool charges cannot be reconstructed from session totals; output already includes reasoning",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "Official OpenAI pricing checked 2026-09-16; Luna input/cache/write/output=0.20/0.02/0.25/1.20 USD per million; https://developers.openai.com/api/docs/pricing",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/ac8db5465e7efaef62b38376760a9e70c459f99f/COST-pranjal-eval155.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
            "notes": "OpenAI's July 9, 2026 changelog announces public availability of GPT-5.6 Sol and Luna. The legacy June 26 date is not the public release date. Proprietary hosted model.",
            "model": "GPT-5.6 Luna",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_155_7fbf62"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/ac8db5465e7efaef62b38376760a9e70c459f99f/pranjal-gpt56luna-codex-energy50-high-500steps-no-preload-modal-v1-155.zip"
          ],
          "task_count": 50,
          "steps_per_task": 15.58,
          "tool_calls_per_task": 39.8,
          "model_responses_per_task": null,
          "variant": "Codex",
          "series": "GPT-5.6 Luna · Codex",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "codex_cli",
            "benchmark_content_hash": "d1c0140e737122942f3108b2d7b73e2c5f894fca048dc22139e78c5772047237",
            "track_name": "api-cpu-agents1-no-preload",
            "measurement_hash": "b61a80d6542471507bf5a8cd8e57db8533379f4232a7cbab3eebf468daccf014",
            "eval_algorithm": "shared-agent-no-preload@1",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "normal_io",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 30,
          "entry_name": "Kimi K3 · high · svc_144_1f26e4",
          "model": "Kimi K3",
          "effort": "high",
          "effort_rank": 3,
          "model_type": "open",
          "performance": 0.796190436008581,
          "time_per_task_sec": 358.12828491998,
          "average_time_per_task_sec": 358.12828491998,
          "median_time_per_task_sec": 159.82407992449998,
          "cost_usd": 0.38225933399999995,
          "turns_per_task": 9.2,
          "release_date": "2026-07-16",
          "average_output_tokens": 774.4086956521741,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_144_1f26e4",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9d7e011c785d433d2329ec51785aa3b6ebbee312/pranjal-kimi-k3-parallel-energy50-high-100steps-modal-v1-144.zip",
          "cost_basis": "recorded_provider_usage_cost",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "d8d4be11925d35f0b29e0e0a4236abceaf3d265d5009a6c319a99b0b015fd367",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.38225933399999995,
          "cost_kind": "recorded",
          "cost_label": "Recorded usage",
          "cost_note": "recorded_provider_usage_cost. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "recorded_provider_usage_cost",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9d7e011c785d433d2329ec51785aa3b6ebbee312/pranjal-kimi-k3-parallel-energy50-high-100steps-modal-v1-144.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "open",
            "release_date": "2026-07-16",
            "release_source": "https://www.kimi.com/en/blog/kimi-k3",
            "access_source": "https://huggingface.co/moonshotai/Kimi-K3",
            "notes": "Public model launch July 16; downloadable weights followed later. API use in the evaluation does not make this a closed-weight model.",
            "model": "Kimi K3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_144_1f26e4"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9d7e011c785d433d2329ec51785aa3b6ebbee312/pranjal-kimi-k3-parallel-energy50-high-100steps-modal-v1-144.zip"
          ],
          "task_count": 50,
          "steps_per_task": 8.76,
          "tool_calls_per_task": 60.2,
          "model_responses_per_task": 9.2,
          "variant": "Batched tool calls",
          "series": "Kimi K3 · Batched tool calls",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "kimi_parallel",
            "benchmark_content_hash": "d1c0140e737122942f3108b2d7b73e2c5f894fca048dc22139e78c5772047237",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "34761619b59911dddf1748656f82a27a24896a268da28e7192a6407cad0f67b7",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 44,
          "entry_name": "Meta Muse Spark 1.1 · xhigh · svc_69_41f88a",
          "model": "Meta Muse Spark 1.1",
          "effort": "xhigh",
          "effort_rank": 4,
          "model_type": "closed",
          "performance": 0.796190436008581,
          "time_per_task_sec": 417.16428902994,
          "average_time_per_task_sec": 417.16428902994,
          "median_time_per_task_sec": 181.350543707,
          "cost_usd": 1.6849374849999998,
          "turns_per_task": 26.02,
          "release_date": "2026-07-09",
          "average_output_tokens": 411.2459646425827,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_69_41f88a",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark11-energy50-xhigh-69.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "979ef3500172b8a8a86b6a5d6d00ffebc4c536d0ff157775f383cb259b5f5a05",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "standard_rate_proxy",
          "cost_label": "Standard-price proxy · no cache",
          "cost_note": "Standard API price proxy, not the archived contributor-tier charge; historical contributor pricing is unverified. Recorded input/output tokens; no input caching assumed. Rates verified 2026-09-16; excludes other charges.",
          "cost_range_min_usd": 0.24221281699999997,
          "cost_range_max_usd": 1.6849374849999998,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://openrouter.ai/api/v1/models (verified 2026-09-16)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.1",
          "cost_unit_prices": {
            "model_id": "meta/muse-spark-1.1",
            "input": 1.25,
            "cached_input": 0.15,
            "output": 4.25,
            "estimate_from_usage": true,
            "source_url": "https://openrouter.ai/api/v1/models",
            "source_page": "https://openrouter.ai/meta/muse-spark-1.1",
            "note": "Standard API price proxy, not the archived contributor-tier charge; historical contributor pricing is unverified."
          },
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "access_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "notes": "Proprietary Muse Spark model served through Meta Model API; not an open-weight Llama or Muse Glimmer release.",
            "model": "Meta Muse Spark 1.1",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_69_41f88a"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark11-energy50-xhigh-69.zip"
          ],
          "task_count": 50,
          "steps_per_task": 69.82,
          "tool_calls_per_task": 145.56,
          "model_responses_per_task": 26.02,
          "variant": "",
          "series": "Meta Muse Spark 1.1",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "meta",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 132,
          "entry_name": "GPT-5.6 Luna · xhigh · svc_156_2bc119",
          "model": "GPT-5.6 Luna",
          "effort": "xhigh",
          "effort_rank": 4,
          "model_type": "closed",
          "performance": 0.776190436008581,
          "time_per_task_sec": 297.23008968142,
          "average_time_per_task_sec": 297.23008968142,
          "median_time_per_task_sec": 177.97039428800002,
          "cost_usd": 0.0631331888,
          "turns_per_task": null,
          "release_date": "2026-07-09",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_156_2bc119",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/ac8db5465e7efaef62b38376760a9e70c459f99f/pranjal-gpt56luna-codex-energy50-xhigh-500steps-no-preload-modal-v1-156.zip",
          "cost_basis": "Luna_standard_short_context_API_equivalent_not_subscription_bill",
          "notes": "Added 2026-09-16 at owner request. GPT-5.6 Luna via Codex, not the direct-API gpt54 template. All 50 original tasks included, including failures. No rerun or regrading. No-preload track/measurement contract; not a same-track ranking against older preload runs. Costs are standard short-context API-equivalent estimates using official 2026-09-16 Luna rates, not subscription charges. All 50 session token summaries complete; cached input is a subset of input, reasoning is a subset of output, recorded cache-write counters are zero. Full CLI model-response/tool trace remains unavailable; observed CLI tool items and harness steps are separate metrics. No additional retained attempts found. Submitted agent.py SHA-256=306fe8c02f9c85a5ff90be95d8efef897a3555ab37fe4f46e84dad68dad0058a. Full per-task normalized evidence and usage breakdowns are linked in companion_cost_usage_url.",
          "quality_flags": "{\"Luna_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":50,\"codex_model_response_count_and_full_tool_calls_unavailable\":50}",
          "selected_task_seed_set_sha256": "688067c69815a2f5cbde4ab5658a6098927ffc259b9e008d73d6d73880589914",
          "run_terminal_events": "[\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 0.0631331888,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Luna_standard_short_context_API_equivalent_not_subscription_bill. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges; assumes standard short-context rates; per-request long-context/fast-mode adjustments and tool charges cannot be reconstructed from session totals; output already includes reasoning",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "Official OpenAI pricing checked 2026-09-16; Luna input/cache/write/output=0.20/0.02/0.25/1.20 USD per million; https://developers.openai.com/api/docs/pricing",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/ac8db5465e7efaef62b38376760a9e70c459f99f/COST-pranjal-eval156.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
            "notes": "OpenAI's July 9, 2026 changelog announces public availability of GPT-5.6 Sol and Luna. The legacy June 26 date is not the public release date. Proprietary hosted model.",
            "model": "GPT-5.6 Luna",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_156_2bc119"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/ac8db5465e7efaef62b38376760a9e70c459f99f/pranjal-gpt56luna-codex-energy50-xhigh-500steps-no-preload-modal-v1-156.zip"
          ],
          "task_count": 50,
          "steps_per_task": 13.88,
          "tool_calls_per_task": 33.1,
          "model_responses_per_task": null,
          "variant": "Codex",
          "series": "GPT-5.6 Luna · Codex",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "codex_cli",
            "benchmark_content_hash": "d1c0140e737122942f3108b2d7b73e2c5f894fca048dc22139e78c5772047237",
            "track_name": "api-cpu-agents1-no-preload",
            "measurement_hash": "b61a80d6542471507bf5a8cd8e57db8533379f4232a7cbab3eebf468daccf014",
            "eval_algorithm": "shared-agent-no-preload@1",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "normal_io",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 133,
          "entry_name": "Yutori n2 · low · svc_11_0dddb1",
          "model": "Yutori n2",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.776190436008581,
          "time_per_task_sec": 331.04671879148003,
          "average_time_per_task_sec": 331.04671879148003,
          "median_time_per_task_sec": 128.622007404,
          "cost_usd": 0.156928266,
          "turns_per_task": 25.32,
          "release_date": "2026-08-26",
          "average_output_tokens": 481.3775671406003,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_11_0dddb1",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61a43287ed7d5eb5aa625e0daa11157dbb5004e1/pranjal-yutori-n2-energy50-low-500steps-no-preload-modal-v1-11.zip",
          "cost_basis": "published_API_rate_estimate_from_logged_usage_and_billed_cache_not_recorded_charge",
          "notes": "Published HF PR23, merged before inventory. All 50 final tasks complete; original no-preload contract and GUI-only tool configuration retained. Full response usage recovered from agent.stdout despite the published README marking usage unavailable; duplicated SDK step usage is not counted again. Cost is an API-rate estimate using logged input/output and billed cached input at publisher rates verified September 19, not a recorded charge. Reasoning-token breakdown is incomplete and remains unknown.",
          "quality_flags": "{\"reasoning_token_breakdown_not_logged_for_all_responses\":50}",
          "selected_task_seed_set_sha256": "15f59bc88045ed6b029cb0b0ba261a3ec7ab083906c56cb8d20f3c55451adb5c",
          "run_terminal_events": "[\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 0.156928266,
          "cost_kind": "source_estimate",
          "cost_label": "API estimate · logged usage",
          "cost_note": "published_API_rate_estimate_from_logged_usage_and_billed_cache_not_recorded_charge. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://yutori.com/blog/introducing-n2 verified 2026-09-19; input/cached input/output=0.50/0.05/4.00 USD per million",
          "cost_source_url": "https://yutori.com/blog/introducing-n2",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "verified_on": "2026-09-19",
            "model_type": "closed",
            "release_date": "2026-08-26",
            "release_source": "https://yutori.com/blog/introducing-n2",
            "access_source": "https://docs.yutori.com/reference/n2",
            "notes": "Publisher's August 26 launch announces Navigator n2 through the Yutori API. Archived response.model is n2; the evaluated GUI-only SDK configuration disables bash/read/write/edit. Sources checked September 19, 2026.",
            "model": "Yutori n2"
          },
          "source_run_ids": [
            "svc_11_0dddb1"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61a43287ed7d5eb5aa625e0daa11157dbb5004e1/pranjal-yutori-n2-energy50-low-500steps-no-preload-modal-v1-11.zip"
          ],
          "task_count": 50,
          "steps_per_task": 22.96,
          "tool_calls_per_task": 58.72,
          "model_responses_per_task": 25.32,
          "variant": "",
          "series": "Yutori n2",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "yutori_n2",
            "benchmark_content_hash": "fc402fbaf9f4a14426ee72a7af3ca8b8565b48cf92d52ac53ad19f5df0772876",
            "track_name": "api-cpu-agents1-no-preload",
            "measurement_hash": "999eacc0929440c65c20fb3785b102d69e756a52098e80f742f769e237bb2996",
            "eval_algorithm": "shared-agent-no-preload@1",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 57,
          "entry_name": "Meta Muse Spark 1.3 · xhigh · svc_68 + rerun3_xhigh13",
          "model": "Meta Muse Spark 1.3",
          "effort": "xhigh",
          "effort_rank": 4,
          "model_type": "closed",
          "performance": 0.776190436008581,
          "time_per_task_sec": 612.07638625182,
          "average_time_per_task_sec": 612.07638625182,
          "median_time_per_task_sec": 368.3002541610001,
          "cost_usd": 0.160709874,
          "turns_per_task": 30.62,
          "release_date": "2026-09-02",
          "average_output_tokens": 375.74853037230565,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_68 + rerun3_xhigh13",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-energy50-xhigh-68.zip | https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-energy50-xhigh-68-rerun3.zip",
          "cost_basis": "unavailable",
          "notes": "Published STITCH-musespark13-xhigh-68.txt: 47 original tasks plus three rerun replacements; overlaps both components and is not an independent run. Published 664.79 seconds/task is mean(agent_wall_sec + env_boot_sec)=664.78846; timed task clock=612.07638625182 seconds/task. Published as one combined 50-task result. Original tasks ran at concurrency 8; the three recovery tasks ran at concurrency 3. Both source archives are retained below.",
          "quality_flags": "{\"result_json_absent_metrics_reduced_from_original_runlog_events\":47}",
          "selected_task_seed_set_sha256": "6ecb8962fa8cc1833813d73b9fc995a7e9153fbe127d8092f45d6c8b7cb50b61",
          "run_terminal_events": "[\"run_failed\"]",
          "row_type": "published_stitch",
          "source_cost_usd": null,
          "cost_kind": "reference_estimate",
          "cost_label": "API estimate · no cache",
          "cost_note": "API-equivalent estimate using recorded tokens and no input caching; not historical billing. Current OpenRouter public API-equivalent token price interval, not historical billing; lower assumes all input cached, upper none; excludes other charges",
          "cost_range_min_usd": 0.0054692598,
          "cost_range_max_usd": 0.160709874,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://openrouter.ai/api/v1/models (snapshot 2026-09-13T17:46:52Z)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.3-contributor",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-02",
            "release_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "access_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "notes": "Proprietary model available through Muse Code and Meta Model API.",
            "model": "Meta Muse Spark 1.3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_68_bca75f",
            "rerun3_xhigh13"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-energy50-xhigh-68.zip",
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-energy50-xhigh-68-rerun3.zip"
          ],
          "task_count": 50,
          "steps_per_task": 56.58,
          "tool_calls_per_task": 86.72,
          "model_responses_per_task": 30.62,
          "variant": "",
          "series": "Meta Muse Spark 1.3",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "meta",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8 | 3"
          }
        },
        {
          "entry_id": 53,
          "entry_name": "Meta Muse Spark 1.3 · medium · svc_65_46a645",
          "model": "Meta Muse Spark 1.3",
          "effort": "medium",
          "effort_rank": 2,
          "model_type": "closed",
          "performance": 0.7754062782047089,
          "time_per_task_sec": 397.56914800776,
          "average_time_per_task_sec": 397.56914800776,
          "median_time_per_task_sec": 251.644451296,
          "cost_usd": 0.125840534,
          "turns_per_task": 26.1,
          "release_date": "2026-09-02",
          "average_output_tokens": 343.00766283524905,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_65_46a645",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-energy50-medium-65.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "33c2bd5ff8ce6ac9d2421225cde340e28c732e03604659cb0ca55019ee8326c2",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "reference_estimate",
          "cost_label": "API estimate · no cache",
          "cost_note": "API-equivalent estimate using recorded tokens and no input caching; not historical billing. Current OpenRouter public API-equivalent token price interval, not historical billing; lower assumes all input cached, upper none; excludes other charges",
          "cost_range_min_usd": 0.00427150068,
          "cost_range_max_usd": 0.125840534,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://openrouter.ai/api/v1/models (snapshot 2026-09-13T17:46:52Z)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.3-contributor",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-02",
            "release_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "access_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "notes": "Proprietary model available through Muse Code and Meta Model API.",
            "model": "Meta Muse Spark 1.3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_65_46a645"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-energy50-medium-65.zip"
          ],
          "task_count": 50,
          "steps_per_task": 41.54,
          "tool_calls_per_task": 69.32,
          "model_responses_per_task": 26.1,
          "variant": "",
          "series": "Meta Muse Spark 1.3",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "meta",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 12,
          "entry_name": "Claude Sonnet 5 · low · svc_25_adbcee",
          "model": "Claude Sonnet 5",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.7580605846606291,
          "time_per_task_sec": 129.43544399922,
          "average_time_per_task_sec": 129.43544399922,
          "median_time_per_task_sec": 56.42313103949999,
          "cost_usd": 0.294072336,
          "turns_per_task": 24.16,
          "release_date": "2026-06-30",
          "average_output_tokens": 143.52649006622516,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_25_adbcee",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-claude-sonnet5-low-osworld-energy50-25.zip",
          "cost_basis": "Claude_standard_global_list_price_estimate_from_logged_usage",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "48497ed8d423e092661c6b8a1b5efe43b325d6329a54fd2be32b5100ccfc22eb",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.294072336,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Claude_standard_global_list_price_estimate_from_logged_usage. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "HF PR16 price snapshot checked 2026-09-13; Opus5 5/6.25/.5/25; Sonnet5 2/2.5/.2/10 USD per million input/5m-write/read/output",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/discussions/16",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-06-30",
            "release_source": "https://www.anthropic.com/news/claude-sonnet-5",
            "access_source": "https://www.anthropic.com/news/claude-sonnet-5",
            "notes": "Proprietary Claude model, available through Anthropic and hosted partners.",
            "model": "Claude Sonnet 5",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_25_adbcee"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-claude-sonnet5-low-osworld-energy50-25.zip"
          ],
          "task_count": 50,
          "steps_per_task": 21.24,
          "tool_calls_per_task": 29.5,
          "model_responses_per_task": 24.16,
          "variant": "",
          "series": "Claude Sonnet 5",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "claude",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "50"
          }
        },
        {
          "entry_id": 129,
          "entry_name": "GPT-5.6 Luna · low · svc_153_795ea7",
          "model": "GPT-5.6 Luna",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.7561904360085809,
          "time_per_task_sec": 144.77467869182001,
          "average_time_per_task_sec": 144.77467869182001,
          "median_time_per_task_sec": 106.5458184545,
          "cost_usd": 0.025388204,
          "turns_per_task": null,
          "release_date": "2026-07-09",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_153_795ea7",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/ac8db5465e7efaef62b38376760a9e70c459f99f/pranjal-gpt56luna-codex-energy50-low-500steps-no-preload-modal-v1-153.zip",
          "cost_basis": "Luna_standard_short_context_API_equivalent_not_subscription_bill",
          "notes": "Added 2026-09-16 at owner request. GPT-5.6 Luna via Codex, not the direct-API gpt54 template. All 50 original tasks included, including failures. No rerun or regrading. No-preload track/measurement contract; not a same-track ranking against older preload runs. Costs are standard short-context API-equivalent estimates using official 2026-09-16 Luna rates, not subscription charges. All 50 session token summaries complete; cached input is a subset of input, reasoning is a subset of output, recorded cache-write counters are zero. Full CLI model-response/tool trace remains unavailable; observed CLI tool items and harness steps are separate metrics. No additional retained attempts found. Submitted agent.py SHA-256=306fe8c02f9c85a5ff90be95d8efef897a3555ab37fe4f46e84dad68dad0058a. Full per-task normalized evidence and usage breakdowns are linked in companion_cost_usage_url.",
          "quality_flags": "{\"Luna_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":50,\"codex_model_response_count_and_full_tool_calls_unavailable\":50}",
          "selected_task_seed_set_sha256": "2b6b3cfc5172a040adbf5630c36f696c456e489f9e74fefdfcaa381a9ad1bc50",
          "run_terminal_events": "[\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 0.025388204,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Luna_standard_short_context_API_equivalent_not_subscription_bill. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges; assumes standard short-context rates; per-request long-context/fast-mode adjustments and tool charges cannot be reconstructed from session totals; output already includes reasoning",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "Official OpenAI pricing checked 2026-09-16; Luna input/cache/write/output=0.20/0.02/0.25/1.20 USD per million; https://developers.openai.com/api/docs/pricing",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/ac8db5465e7efaef62b38376760a9e70c459f99f/COST-pranjal-eval153.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
            "notes": "OpenAI's July 9, 2026 changelog announces public availability of GPT-5.6 Sol and Luna. The legacy June 26 date is not the public release date. Proprietary hosted model.",
            "model": "GPT-5.6 Luna",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_153_795ea7"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/ac8db5465e7efaef62b38376760a9e70c459f99f/pranjal-gpt56luna-codex-energy50-low-500steps-no-preload-modal-v1-153.zip"
          ],
          "task_count": 50,
          "steps_per_task": 7.22,
          "tool_calls_per_task": 26.44,
          "model_responses_per_task": null,
          "variant": "Codex",
          "series": "GPT-5.6 Luna · Codex",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "codex_cli",
            "benchmark_content_hash": "d1c0140e737122942f3108b2d7b73e2c5f894fca048dc22139e78c5772047237",
            "track_name": "api-cpu-agents1-no-preload",
            "measurement_hash": "b61a80d6542471507bf5a8cd8e57db8533379f4232a7cbab3eebf468daccf014",
            "eval_algorithm": "shared-agent-no-preload@1",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "normal_io",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 58,
          "entry_name": "MiniMax M3 · thinking_off · svc_51_121311",
          "model": "MiniMax M3",
          "effort": "thinking_off",
          "effort_rank": 0,
          "model_type": "open",
          "performance": 0.7561904360085809,
          "time_per_task_sec": 253.80420472314,
          "average_time_per_task_sec": 253.80420472314,
          "median_time_per_task_sec": 128.15256609600002,
          "cost_usd": null,
          "turns_per_task": 35.14,
          "release_date": "2026-06-01",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_51_121311",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-minimax-m3-energy50-thinking-off-51.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{\"output_truncated_at_400_characters_and_usage_not_logged\":50}",
          "selected_task_seed_set_sha256": "8ed2b651de4dc09d31c77216bdad82bc5749932e8563b357940e34d8ba868c4f",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "unavailable",
          "cost_label": "Usage not logged",
          "cost_note": "Standard API rates through 512K context; above 512K, input/output/cache rates double. Priority pricing differs. Archived responses do not retain token usage, so unit prices cannot yield a per-task total.",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 0,
          "cost_total_tasks": 50,
          "cost_source": "https://platform.minimax.io/subscribe/token-plan?tab=api-enterprise (verified 2026-09-16)",
          "cost_source_url": "https://platform.minimax.io/subscribe/token-plan?tab=api-enterprise",
          "cost_pricing_model": "minimax/minimax-m3",
          "cost_unit_prices": {
            "model_id": "minimax/minimax-m3",
            "input": 0.3,
            "cached_input": 0.06,
            "output": 1.2,
            "estimate_from_usage": false,
            "unit_note": "standard · ≤512K context",
            "source_url": "https://platform.minimax.io/subscribe/token-plan?tab=api-enterprise",
            "source_page": "https://www.minimax.io/blog/minimax-m3",
            "note": "Standard API rates through 512K context; above 512K, input/output/cache rates double. Priority pricing differs. Archived responses do not retain token usage, so unit prices cannot yield a per-task total."
          },
          "model_metadata": {
            "model_type": "open",
            "release_date": "2026-06-01",
            "release_source": "https://www.minimax.io/blog/minimax-m3",
            "access_source": "https://huggingface.co/MiniMaxAI/MiniMax-M3",
            "notes": "Public model launch June 1; downloadable weights followed later under the MiniMax Community License. The hyphenated legacy label identifies the same model.",
            "model": "MiniMax M3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_51_121311"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-minimax-m3-energy50-thinking-off-51.zip"
          ],
          "task_count": 50,
          "steps_per_task": 34.32,
          "tool_calls_per_task": 51.1,
          "model_responses_per_task": 35.14,
          "variant": "",
          "series": "MiniMax M3",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "minimax_m3",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 135,
          "entry_name": "Yutori n2 · none · svc_10_cd0ed2",
          "model": "Yutori n2",
          "effort": "none",
          "effort_rank": -1,
          "model_type": "closed",
          "performance": 0.7561904360085809,
          "time_per_task_sec": 298.49816499184,
          "average_time_per_task_sec": 298.49816499184,
          "median_time_per_task_sec": 195.89178775,
          "cost_usd": 0.081063778,
          "turns_per_task": 20.9,
          "release_date": "2026-08-26",
          "average_output_tokens": 174.73588516746412,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_10_cd0ed2",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61a43287ed7d5eb5aa625e0daa11157dbb5004e1/pranjal-yutori-n2-energy50-none-500steps-no-preload-modal-v1-10.zip",
          "cost_basis": "published_API_rate_estimate_from_logged_usage_and_billed_cache_not_recorded_charge",
          "notes": "Published HF PR23, merged before inventory. All 50 final tasks complete; original no-preload contract and GUI-only tool configuration retained. Full response usage recovered from agent.stdout despite the published README marking usage unavailable; duplicated SDK step usage is not counted again. Cost is an API-rate estimate using logged input/output and billed cached input at publisher rates verified September 19, not a recorded charge. Reasoning-token breakdown is incomplete and remains unknown.",
          "quality_flags": "{\"reasoning_token_breakdown_not_logged_for_all_responses\":50}",
          "selected_task_seed_set_sha256": "a1f2bedaf414b6d7a8ed97e0162a92301725ddfd9c7f3a09d160b6043c7431eb",
          "run_terminal_events": "[\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 0.081063778,
          "cost_kind": "source_estimate",
          "cost_label": "API estimate · logged usage",
          "cost_note": "published_API_rate_estimate_from_logged_usage_and_billed_cache_not_recorded_charge. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://yutori.com/blog/introducing-n2 verified 2026-09-19; input/cached input/output=0.50/0.05/4.00 USD per million",
          "cost_source_url": "https://yutori.com/blog/introducing-n2",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "verified_on": "2026-09-19",
            "model_type": "closed",
            "release_date": "2026-08-26",
            "release_source": "https://yutori.com/blog/introducing-n2",
            "access_source": "https://docs.yutori.com/reference/n2",
            "notes": "Publisher's August 26 launch announces Navigator n2 through the Yutori API. Archived response.model is n2; the evaluated GUI-only SDK configuration disables bash/read/write/edit. Sources checked September 19, 2026.",
            "model": "Yutori n2"
          },
          "source_run_ids": [
            "svc_10_cd0ed2"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61a43287ed7d5eb5aa625e0daa11157dbb5004e1/pranjal-yutori-n2-energy50-none-500steps-no-preload-modal-v1-10.zip"
          ],
          "task_count": 50,
          "steps_per_task": 18.74,
          "tool_calls_per_task": 56.54,
          "model_responses_per_task": 20.9,
          "variant": "",
          "series": "Yutori n2",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "yutori_n2",
            "benchmark_content_hash": "fc402fbaf9f4a14426ee72a7af3ca8b8565b48cf92d52ac53ad19f5df0772876",
            "track_name": "api-cpu-agents1-no-preload",
            "measurement_hash": "999eacc0929440c65c20fb3785b102d69e756a52098e80f742f769e237bb2996",
            "eval_algorithm": "shared-agent-no-preload@1",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 16,
          "entry_name": "GPT-5.6 Luna · xhigh · svc_21_00be0a",
          "model": "GPT-5.6 Luna",
          "effort": "xhigh",
          "effort_rank": 4,
          "model_type": "closed",
          "performance": 0.7361904360085809,
          "time_per_task_sec": 135.31637142254,
          "average_time_per_task_sec": 135.31637142254,
          "median_time_per_task_sec": 60.4715172695,
          "cost_usd": 0.0385505,
          "turns_per_task": 18.96,
          "release_date": "2026-07-09",
          "average_output_tokens": 211.62552742616035,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_21_00be0a",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gpt56luna-xhigh-osworld-energy50-21.zip",
          "cost_basis": "sum_of_rounded_per_response_template_estimates",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "4155769733ceae8b4b1e91f75da30c49ad2ae06f81d37bab6f9d225e5d5b5520",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.0385505,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "sum_of_rounded_per_response_template_estimates. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "sum_of_rounded_per_response_template_estimates",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gpt56luna-xhigh-osworld-energy50-21.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
            "notes": "OpenAI's July 9, 2026 changelog announces public availability of GPT-5.6 Sol and Luna. The legacy June 26 date is not the public release date. Proprietary hosted model.",
            "model": "GPT-5.6 Luna",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_21_00be0a"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gpt56luna-xhigh-osworld-energy50-21.zip"
          ],
          "task_count": 50,
          "steps_per_task": 18.02,
          "tool_calls_per_task": 47.48,
          "model_responses_per_task": 18.96,
          "variant": "Direct API",
          "series": "GPT-5.6 Luna · Direct API",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "gpt54",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "50"
          }
        },
        {
          "entry_id": 32,
          "entry_name": "Kimi K3 · low · svc_143_e8d33d",
          "model": "Kimi K3",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "open",
          "performance": 0.7361904360085809,
          "time_per_task_sec": 149.49769912932,
          "average_time_per_task_sec": 149.49769912932,
          "median_time_per_task_sec": 102.31283380049999,
          "cost_usd": 0.174015144,
          "turns_per_task": 6.9,
          "release_date": "2026-07-16",
          "average_output_tokens": 270.2782608695652,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_143_e8d33d",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9d7e011c785d433d2329ec51785aa3b6ebbee312/pranjal-kimi-k3-parallel-energy50-low-100steps-modal-v1-143.zip",
          "cost_basis": "recorded_provider_usage_cost",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "375bd7ee37d05fb7287315c5cdaf14ae50a60134f287ded1144c593e2022fec3",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.174015144,
          "cost_kind": "recorded",
          "cost_label": "Recorded usage",
          "cost_note": "recorded_provider_usage_cost. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "recorded_provider_usage_cost",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9d7e011c785d433d2329ec51785aa3b6ebbee312/pranjal-kimi-k3-parallel-energy50-low-100steps-modal-v1-143.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "open",
            "release_date": "2026-07-16",
            "release_source": "https://www.kimi.com/en/blog/kimi-k3",
            "access_source": "https://huggingface.co/moonshotai/Kimi-K3",
            "notes": "Public model launch July 16; downloadable weights followed later. API use in the evaluation does not make this a closed-weight model.",
            "model": "Kimi K3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_143_e8d33d"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9d7e011c785d433d2329ec51785aa3b6ebbee312/pranjal-kimi-k3-parallel-energy50-low-100steps-modal-v1-143.zip"
          ],
          "task_count": 50,
          "steps_per_task": 5.74,
          "tool_calls_per_task": 30.0,
          "model_responses_per_task": 6.9,
          "variant": "Batched tool calls",
          "series": "Kimi K3 · Batched tool calls",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "kimi_parallel",
            "benchmark_content_hash": "d1c0140e737122942f3108b2d7b73e2c5f894fca048dc22139e78c5772047237",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "34761619b59911dddf1748656f82a27a24896a268da28e7192a6407cad0f67b7",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 51,
          "entry_name": "Meta Muse Spark 1.3 · high · svc_70_8f6cda",
          "model": "Meta Muse Spark 1.3",
          "effort": "high",
          "effort_rank": 3,
          "model_type": "closed",
          "performance": 0.7361904360085809,
          "time_per_task_sec": 506.90729824484,
          "average_time_per_task_sec": 506.90729824484,
          "median_time_per_task_sec": 246.90281670399992,
          "cost_usd": 0.14786779,
          "turns_per_task": 28.98,
          "release_date": "2026-09-02",
          "average_output_tokens": 363.6204278812974,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_70_8f6cda",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-energy50-high-70.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "b572502094cca9e6e1ae2ed27a547f4d9197421d4e87e861615691440111b594",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "reference_estimate",
          "cost_label": "API estimate · no cache",
          "cost_note": "API-equivalent estimate using recorded tokens and no input caching; not historical billing. Current OpenRouter public API-equivalent token price interval, not historical billing; lower assumes all input cached, upper none; excludes other charges",
          "cost_range_min_usd": 0.005022748919999999,
          "cost_range_max_usd": 0.14786779,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://openrouter.ai/api/v1/models (snapshot 2026-09-13T17:46:52Z)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.3-contributor",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-02",
            "release_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "access_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "notes": "Proprietary model available through Muse Code and Meta Model API.",
            "model": "Meta Muse Spark 1.3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_70_8f6cda"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-energy50-high-70.zip"
          ],
          "task_count": 50,
          "steps_per_task": 51.3,
          "tool_calls_per_task": 84.78,
          "model_responses_per_task": 28.98,
          "variant": "",
          "series": "Meta Muse Spark 1.3",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "meta",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 31,
          "entry_name": "Kimi K3 · low · svc_89_b4886a",
          "model": "Kimi K3",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "open",
          "performance": 0.7161904360085809,
          "time_per_task_sec": 176.6093889107,
          "average_time_per_task_sec": 176.6093889107,
          "median_time_per_task_sec": 84.13998422899999,
          "cost_usd": 0.280166118,
          "turns_per_task": 9.62,
          "release_date": "2026-07-16",
          "average_output_tokens": 329.5966735966736,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_89_b4886a",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-kimi-k3-energy50-low-modal-v1-89.zip",
          "cost_basis": "recorded_provider_usage_cost",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "34f3e78a07b03abf26d28ce98f0b62900846e18370cc542f38c43a84bba2c5d4",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.280166118,
          "cost_kind": "recorded",
          "cost_label": "Recorded usage",
          "cost_note": "recorded_provider_usage_cost. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "recorded_provider_usage_cost",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-kimi-k3-energy50-low-modal-v1-89.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "open",
            "release_date": "2026-07-16",
            "release_source": "https://www.kimi.com/en/blog/kimi-k3",
            "access_source": "https://huggingface.co/moonshotai/Kimi-K3",
            "notes": "Public model launch July 16; downloadable weights followed later. API use in the evaluation does not make this a closed-weight model.",
            "model": "Kimi K3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_89_b4886a"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-kimi-k3-energy50-low-modal-v1-89.zip"
          ],
          "task_count": 50,
          "steps_per_task": 8.12,
          "tool_calls_per_task": 55.1,
          "model_responses_per_task": 9.62,
          "variant": "Single tool call",
          "series": "Kimi K3 · Single tool call",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "kimi_k3",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 59,
          "entry_name": "MiniMax M3 · thinking_on · svc_52_554434",
          "model": "MiniMax M3",
          "effort": "thinking_on",
          "effort_rank": 6,
          "model_type": "open",
          "performance": 0.6961904360085809,
          "time_per_task_sec": 345.1228296594,
          "average_time_per_task_sec": 345.1228296594,
          "median_time_per_task_sec": 183.42030929049997,
          "cost_usd": null,
          "turns_per_task": 35.94,
          "release_date": "2026-06-01",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_52_554434",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-minimax-m3-energy50-thinking-on-52.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{\"output_truncated_at_400_characters_and_usage_not_logged\":50}",
          "selected_task_seed_set_sha256": "61555ca724cb1d535f57603f64af271095cec4e1e2927eff5a2a324ceef5925f",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "unavailable",
          "cost_label": "Usage not logged",
          "cost_note": "Standard API rates through 512K context; above 512K, input/output/cache rates double. Priority pricing differs. Archived responses do not retain token usage, so unit prices cannot yield a per-task total.",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 0,
          "cost_total_tasks": 50,
          "cost_source": "https://platform.minimax.io/subscribe/token-plan?tab=api-enterprise (verified 2026-09-16)",
          "cost_source_url": "https://platform.minimax.io/subscribe/token-plan?tab=api-enterprise",
          "cost_pricing_model": "minimax/minimax-m3",
          "cost_unit_prices": {
            "model_id": "minimax/minimax-m3",
            "input": 0.3,
            "cached_input": 0.06,
            "output": 1.2,
            "estimate_from_usage": false,
            "unit_note": "standard · ≤512K context",
            "source_url": "https://platform.minimax.io/subscribe/token-plan?tab=api-enterprise",
            "source_page": "https://www.minimax.io/blog/minimax-m3",
            "note": "Standard API rates through 512K context; above 512K, input/output/cache rates double. Priority pricing differs. Archived responses do not retain token usage, so unit prices cannot yield a per-task total."
          },
          "model_metadata": {
            "model_type": "open",
            "release_date": "2026-06-01",
            "release_source": "https://www.minimax.io/blog/minimax-m3",
            "access_source": "https://huggingface.co/MiniMaxAI/MiniMax-M3",
            "notes": "Public model launch June 1; downloadable weights followed later under the MiniMax Community License. The hyphenated legacy label identifies the same model.",
            "model": "MiniMax M3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_52_554434"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-minimax-m3-energy50-thinking-on-52.zip"
          ],
          "task_count": 50,
          "steps_per_task": 35.0,
          "tool_calls_per_task": 55.3,
          "model_responses_per_task": 35.94,
          "variant": "",
          "series": "MiniMax M3",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "minimax_m3",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 39,
          "entry_name": "Meta Muse Spark 1.1 · medium · svc_59_d8a6d4",
          "model": "Meta Muse Spark 1.1",
          "effort": "medium",
          "effort_rank": 2,
          "model_type": "closed",
          "performance": 0.6961904360085809,
          "time_per_task_sec": 359.76952606414,
          "average_time_per_task_sec": 359.76952606414,
          "median_time_per_task_sec": 138.19131289249998,
          "cost_usd": 1.3785025299999998,
          "turns_per_task": 22.82,
          "release_date": "2026-07-09",
          "average_output_tokens": 322.15425065731813,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_59_d8a6d4",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark11-energy50-medium-59.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "572f288adf664e3254edaa26282b389479b1dab52e4fd6227c900390fde9b417",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "standard_rate_proxy",
          "cost_label": "Standard-price proxy · no cache",
          "cost_note": "Standard API price proxy, not the archived contributor-tier charge; historical contributor pricing is unverified. Recorded input/output tokens; no input caching assumed. Rates verified 2026-09-16; excludes other charges.",
          "cost_range_min_usd": 0.192915138,
          "cost_range_max_usd": 1.3785025299999998,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://openrouter.ai/api/v1/models (verified 2026-09-16)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.1",
          "cost_unit_prices": {
            "model_id": "meta/muse-spark-1.1",
            "input": 1.25,
            "cached_input": 0.15,
            "output": 4.25,
            "estimate_from_usage": true,
            "source_url": "https://openrouter.ai/api/v1/models",
            "source_page": "https://openrouter.ai/meta/muse-spark-1.1",
            "note": "Standard API price proxy, not the archived contributor-tier charge; historical contributor pricing is unverified."
          },
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "access_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "notes": "Proprietary Muse Spark model served through Meta Model API; not an open-weight Llama or Muse Glimmer release.",
            "model": "Meta Muse Spark 1.1",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_59_d8a6d4"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark11-energy50-medium-59.zip"
          ],
          "task_count": 50,
          "steps_per_task": 54.36,
          "tool_calls_per_task": 114.0,
          "model_responses_per_task": 22.82,
          "variant": "",
          "series": "Meta Muse Spark 1.1",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "meta",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 35,
          "entry_name": "Meta Muse Spark 1.1 · high · svc_58_38520a",
          "model": "Meta Muse Spark 1.1",
          "effort": "high",
          "effort_rank": 3,
          "model_type": "closed",
          "performance": 0.6961904360085809,
          "time_per_task_sec": 368.47019029020004,
          "average_time_per_task_sec": 368.47019029020004,
          "median_time_per_task_sec": 105.71536493200001,
          "cost_usd": 1.7059075799999999,
          "turns_per_task": 25.66,
          "release_date": "2026-07-09",
          "average_output_tokens": 394.8659392049883,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_58_38520a",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark11-energy50-high-58.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "773bcd813e13dbda8aab9d4ec9bebe88171a0a039fc396cdfb9f2f6378b0f53d",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "standard_rate_proxy",
          "cost_label": "Standard-price proxy · no cache",
          "cost_note": "Standard API price proxy, not the archived contributor-tier charge; historical contributor pricing is unverified. Recorded input/output tokens; no input caching assumed. Rates verified 2026-09-16; excludes other charges.",
          "cost_range_min_usd": 0.24260356199999997,
          "cost_range_max_usd": 1.7059075799999999,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://openrouter.ai/api/v1/models (verified 2026-09-16)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.1",
          "cost_unit_prices": {
            "model_id": "meta/muse-spark-1.1",
            "input": 1.25,
            "cached_input": 0.15,
            "output": 4.25,
            "estimate_from_usage": true,
            "source_url": "https://openrouter.ai/api/v1/models",
            "source_page": "https://openrouter.ai/meta/muse-spark-1.1",
            "note": "Standard API price proxy, not the archived contributor-tier charge; historical contributor pricing is unverified."
          },
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "access_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "notes": "Proprietary Muse Spark model served through Meta Model API; not an open-weight Llama or Muse Glimmer release.",
            "model": "Meta Muse Spark 1.1",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_58_38520a"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark11-energy50-high-58.zip"
          ],
          "task_count": 50,
          "steps_per_task": 69.9,
          "tool_calls_per_task": 126.88,
          "model_responses_per_task": 25.66,
          "variant": "",
          "series": "Meta Muse Spark 1.1",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "meta",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 37,
          "entry_name": "Meta Muse Spark 1.1 · low · svc_60_72500b",
          "model": "Meta Muse Spark 1.1",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.6961904360085809,
          "time_per_task_sec": 370.42262176936,
          "average_time_per_task_sec": 370.42262176936,
          "median_time_per_task_sec": 124.57902521600002,
          "cost_usd": 1.311594645,
          "turns_per_task": 22.12,
          "release_date": "2026-07-09",
          "average_output_tokens": 295.0153707052441,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_60_72500b",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark11-energy50-low-60.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "2f2692e863f371097c6ff79fe2946b07d28f343914fc6fd4db8ced35ce41223c",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "standard_rate_proxy",
          "cost_label": "Standard-price proxy · no cache",
          "cost_note": "Standard API price proxy, not the archived contributor-tier charge; historical contributor pricing is unverified. Recorded input/output tokens; no input caching assumed. Rates verified 2026-09-16; excludes other charges.",
          "cost_range_min_usd": 0.18179762499999996,
          "cost_range_max_usd": 1.311594645,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://openrouter.ai/api/v1/models (verified 2026-09-16)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.1",
          "cost_unit_prices": {
            "model_id": "meta/muse-spark-1.1",
            "input": 1.25,
            "cached_input": 0.15,
            "output": 4.25,
            "estimate_from_usage": true,
            "source_url": "https://openrouter.ai/api/v1/models",
            "source_page": "https://openrouter.ai/meta/muse-spark-1.1",
            "note": "Standard API price proxy, not the archived contributor-tier charge; historical contributor pricing is unverified."
          },
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "access_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "notes": "Proprietary Muse Spark model served through Meta Model API; not an open-weight Llama or Muse Glimmer release.",
            "model": "Meta Muse Spark 1.1",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_60_72500b"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark11-energy50-low-60.zip"
          ],
          "task_count": 50,
          "steps_per_task": 58.48,
          "tool_calls_per_task": 149.86,
          "model_responses_per_task": 22.12,
          "variant": "",
          "series": "Meta Muse Spark 1.1",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "meta",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 50,
          "entry_name": "Meta Muse Spark 1.2 · xhigh · svc_39_025394",
          "model": "Meta Muse Spark 1.2",
          "effort": "xhigh",
          "effort_rank": 4,
          "model_type": "closed",
          "performance": 0.6961904360085809,
          "time_per_task_sec": 720.9012414855799,
          "average_time_per_task_sec": 720.9012414855799,
          "median_time_per_task_sec": 520.7893415160001,
          "cost_usd": 0.24448311,
          "turns_per_task": 43.08,
          "release_date": "2026-08-05",
          "average_output_tokens": 399.22191272052,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_39_025394",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark12-energy50-xhigh-39.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "2e76915ad66aa1d5120503c8d9b85082eda851c4a5d83ea54f811ad70bd4fce0",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "reference_estimate",
          "cost_label": "API estimate · no cache",
          "cost_note": "API-equivalent estimate using recorded tokens and no input caching; not historical billing. Current OpenRouter public API-equivalent token price interval, not historical billing; lower assumes all input cached, upper none; excludes other charges",
          "cost_range_min_usd": 0.00826056428,
          "cost_range_max_usd": 0.24448311,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://openrouter.ai/api/v1/models (snapshot 2026-09-13T17:46:52Z)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.2-contributor",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-08-05",
            "release_source": "https://research.meta.ai/blog/introducing-muse-code-and-muse-spark-1-2",
            "access_source": "https://research.meta.ai/blog/introducing-muse-code-and-muse-spark-1-2",
            "notes": "Proprietary model available through Muse Code and Meta Model API.",
            "model": "Meta Muse Spark 1.2",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_39_025394"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark12-energy50-xhigh-39.zip"
          ],
          "task_count": 50,
          "steps_per_task": 66.54,
          "tool_calls_per_task": 200.52,
          "model_responses_per_task": 43.08,
          "variant": "",
          "series": "Meta Muse Spark 1.2",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "meta",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 130,
          "entry_name": "GPT-5.6 Luna · medium · svc_154_323b97",
          "model": "GPT-5.6 Luna",
          "effort": "medium",
          "effort_rank": 2,
          "model_type": "closed",
          "performance": 0.6761904360085809,
          "time_per_task_sec": 188.84819598837998,
          "average_time_per_task_sec": 188.84819598837998,
          "median_time_per_task_sec": 129.3341464855,
          "cost_usd": 0.035304857599999996,
          "turns_per_task": null,
          "release_date": "2026-07-09",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_154_323b97",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/ac8db5465e7efaef62b38376760a9e70c459f99f/pranjal-gpt56luna-codex-energy50-medium-500steps-no-preload-modal-v1-154.zip",
          "cost_basis": "Luna_standard_short_context_API_equivalent_not_subscription_bill",
          "notes": "Added 2026-09-16 at owner request. GPT-5.6 Luna via Codex, not the direct-API gpt54 template. All 50 original tasks included, including failures. No rerun or regrading. No-preload track/measurement contract; not a same-track ranking against older preload runs. Costs are standard short-context API-equivalent estimates using official 2026-09-16 Luna rates, not subscription charges. All 50 session token summaries complete; cached input is a subset of input, reasoning is a subset of output, recorded cache-write counters are zero. Full CLI model-response/tool trace remains unavailable; observed CLI tool items and harness steps are separate metrics. No additional retained attempts found. Submitted agent.py SHA-256=306fe8c02f9c85a5ff90be95d8efef897a3555ab37fe4f46e84dad68dad0058a. One original agent_error on osworld_libreoffice_writer_adf5e2c3-64c7-4644-b7b6-d2f0167927e7/seed_1031319744: the environment HTTPS endpoint refused the /done connection and the gateway failed. Its zero score, timing and recorded usage remain included; not a clean failure-free comparison. Full per-task normalized evidence and usage breakdowns are linked in companion_cost_usage_url.",
          "quality_flags": "{\"Luna_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":50,\"codex_model_response_count_and_full_tool_calls_unavailable\":50}",
          "selected_task_seed_set_sha256": "4870eadccad127c9cef51031baa497ef3d224cbf46ae44ac40f4f4e4340c5105",
          "run_terminal_events": "[\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 0.035304857599999996,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Luna_standard_short_context_API_equivalent_not_subscription_bill. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges; assumes standard short-context rates; per-request long-context/fast-mode adjustments and tool charges cannot be reconstructed from session totals; output already includes reasoning",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "Official OpenAI pricing checked 2026-09-16; Luna input/cache/write/output=0.20/0.02/0.25/1.20 USD per million; https://developers.openai.com/api/docs/pricing",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/ac8db5465e7efaef62b38376760a9e70c459f99f/COST-pranjal-eval154.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
            "notes": "OpenAI's July 9, 2026 changelog announces public availability of GPT-5.6 Sol and Luna. The legacy June 26 date is not the public release date. Proprietary hosted model.",
            "model": "GPT-5.6 Luna",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_154_323b97"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/ac8db5465e7efaef62b38376760a9e70c459f99f/pranjal-gpt56luna-codex-energy50-medium-500steps-no-preload-modal-v1-154.zip"
          ],
          "task_count": 50,
          "steps_per_task": 9.5,
          "tool_calls_per_task": 25.64,
          "model_responses_per_task": null,
          "variant": "Codex",
          "series": "GPT-5.6 Luna · Codex",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "codex_cli",
            "benchmark_content_hash": "d1c0140e737122942f3108b2d7b73e2c5f894fca048dc22139e78c5772047237",
            "track_name": "api-cpu-agents1-no-preload",
            "measurement_hash": "b61a80d6542471507bf5a8cd8e57db8533379f4232a7cbab3eebf468daccf014",
            "eval_algorithm": "shared-agent-no-preload@1",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "normal_io",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 134,
          "entry_name": "Yutori n2 · medium · svc_12_026ed7",
          "model": "Yutori n2",
          "effort": "medium",
          "effort_rank": 2,
          "model_type": "closed",
          "performance": 0.6761904360085809,
          "time_per_task_sec": 265.43294287585996,
          "average_time_per_task_sec": 265.43294287585996,
          "median_time_per_task_sec": 129.628457145,
          "cost_usd": 0.10999661200000001,
          "turns_per_task": 21.12,
          "release_date": "2026-08-26",
          "average_output_tokens": 358.0369318181818,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_12_026ed7",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61a43287ed7d5eb5aa625e0daa11157dbb5004e1/pranjal-yutori-n2-energy50-medium-500steps-no-preload-modal-v1-12.zip",
          "cost_basis": "published_API_rate_estimate_from_logged_usage_and_billed_cache_not_recorded_charge",
          "notes": "Published HF PR23, merged before inventory. All 50 final tasks complete; original no-preload contract and GUI-only tool configuration retained. Full response usage recovered from agent.stdout despite the published README marking usage unavailable; duplicated SDK step usage is not counted again. Cost is an API-rate estimate using logged input/output and billed cached input at publisher rates verified September 19, not a recorded charge. Reasoning-token breakdown is incomplete and remains unknown.",
          "quality_flags": "{\"reasoning_token_breakdown_not_logged_for_all_responses\":50}",
          "selected_task_seed_set_sha256": "58928fd28da9737fd9e9957d61116535bc3dc57999745c72e161340dd2f3e847",
          "run_terminal_events": "[\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 0.10999661200000001,
          "cost_kind": "source_estimate",
          "cost_label": "API estimate · logged usage",
          "cost_note": "published_API_rate_estimate_from_logged_usage_and_billed_cache_not_recorded_charge. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://yutori.com/blog/introducing-n2 verified 2026-09-19; input/cached input/output=0.50/0.05/4.00 USD per million",
          "cost_source_url": "https://yutori.com/blog/introducing-n2",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "verified_on": "2026-09-19",
            "model_type": "closed",
            "release_date": "2026-08-26",
            "release_source": "https://yutori.com/blog/introducing-n2",
            "access_source": "https://docs.yutori.com/reference/n2",
            "notes": "Publisher's August 26 launch announces Navigator n2 through the Yutori API. Archived response.model is n2; the evaluated GUI-only SDK configuration disables bash/read/write/edit. Sources checked September 19, 2026.",
            "model": "Yutori n2"
          },
          "source_run_ids": [
            "svc_12_026ed7"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61a43287ed7d5eb5aa625e0daa11157dbb5004e1/pranjal-yutori-n2-energy50-medium-500steps-no-preload-modal-v1-12.zip"
          ],
          "task_count": 50,
          "steps_per_task": 19.06,
          "tool_calls_per_task": 55.14,
          "model_responses_per_task": 21.12,
          "variant": "",
          "series": "Yutori n2",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "yutori_n2",
            "benchmark_content_hash": "fc402fbaf9f4a14426ee72a7af3ca8b8565b48cf92d52ac53ad19f5df0772876",
            "track_name": "api-cpu-agents1-no-preload",
            "measurement_hash": "999eacc0929440c65c20fb3785b102d69e756a52098e80f742f769e237bb2996",
            "eval_algorithm": "shared-agent-no-preload@1",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 41,
          "entry_name": "Meta Muse Spark 1.1 · minimal · svc_61_96ce3f",
          "model": "Meta Muse Spark 1.1",
          "effort": "minimal",
          "effort_rank": 0,
          "model_type": "closed",
          "performance": 0.6761904360085809,
          "time_per_task_sec": 318.54680623266,
          "average_time_per_task_sec": 318.54680623266,
          "median_time_per_task_sec": 94.0166365645,
          "cost_usd": 1.1840251000000002,
          "turns_per_task": 20.62,
          "release_date": "2026-07-09",
          "average_output_tokens": 288.87972841901063,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_61_96ce3f",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark11-energy50-minimal-61.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "da6f456c817ebf4e1f6669f6fe52d36c90f6f10aa4556341cebd6b35a51faa54",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "standard_rate_proxy",
          "cost_label": "Standard-price proxy · no cache",
          "cost_note": "Standard API price proxy, not the archived contributor-tier charge; historical contributor pricing is unverified. Recorded input/output tokens; no input caching assumed. Rates verified 2026-09-16; excludes other charges.",
          "cost_range_min_usd": 0.16436107,
          "cost_range_max_usd": 1.1840251000000002,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://openrouter.ai/api/v1/models (verified 2026-09-16)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.1",
          "cost_unit_prices": {
            "model_id": "meta/muse-spark-1.1",
            "input": 1.25,
            "cached_input": 0.15,
            "output": 4.25,
            "estimate_from_usage": true,
            "source_url": "https://openrouter.ai/api/v1/models",
            "source_page": "https://openrouter.ai/meta/muse-spark-1.1",
            "note": "Standard API price proxy, not the archived contributor-tier charge; historical contributor pricing is unverified."
          },
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "access_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "notes": "Proprietary Muse Spark model served through Meta Model API; not an open-weight Llama or Muse Glimmer release.",
            "model": "Meta Muse Spark 1.1",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_61_96ce3f"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark11-energy50-minimal-61.zip"
          ],
          "task_count": 50,
          "steps_per_task": 53.44,
          "tool_calls_per_task": 131.68,
          "model_responses_per_task": 20.62,
          "variant": "",
          "series": "Meta Muse Spark 1.1",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "meta",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 52,
          "entry_name": "Meta Muse Spark 1.3 · low · svc_66_660b4a",
          "model": "Meta Muse Spark 1.3",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.6761904360085809,
          "time_per_task_sec": 360.9915694551,
          "average_time_per_task_sec": 360.9915694551,
          "median_time_per_task_sec": 197.21761747899998,
          "cost_usd": 0.107121342,
          "turns_per_task": 23.54,
          "release_date": "2026-09-02",
          "average_output_tokens": 286.62531860662705,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_66_660b4a",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-energy50-low-66.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "baa5ed9366adf109fd961523bb7ab64ff47b134e3258ccc5d9ffa09a51f561c0",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "reference_estimate",
          "cost_label": "API estimate · no cache",
          "cost_note": "API-equivalent estimate using recorded tokens and no input caching; not historical billing. Current OpenRouter public API-equivalent token price interval, not historical billing; lower assumes all input cached, upper none; excludes other charges",
          "cost_range_min_usd": 0.0034648702,
          "cost_range_max_usd": 0.107121342,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://openrouter.ai/api/v1/models (snapshot 2026-09-13T17:46:52Z)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.3-contributor",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-02",
            "release_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "access_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "notes": "Proprietary model available through Muse Code and Meta Model API.",
            "model": "Meta Muse Spark 1.3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_66_660b4a"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-energy50-low-66.zip"
          ],
          "task_count": 50,
          "steps_per_task": 40.5,
          "tool_calls_per_task": 71.82,
          "model_responses_per_task": 23.54,
          "variant": "",
          "series": "Meta Muse Spark 1.3",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "meta",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 127,
          "entry_name": "GPT-5.6 Sol · low · svc_62_9b9e92",
          "model": "GPT-5.6 Sol",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.6581298513479519,
          "time_per_task_sec": 100.09055964158,
          "average_time_per_task_sec": 100.09055964158,
          "median_time_per_task_sec": 78.56391550650001,
          "cost_usd": 0.3190882,
          "turns_per_task": 12.16,
          "release_date": "2026-07-09",
          "average_output_tokens": 101.65789473684211,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_62_9b9e92",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/5d04d3b4b91dceb4c90e8592ce3235c1965c9db2/jykoh-gpt56sol-osworld-energy50-representative-modal-low-500steps-62.zip",
          "cost_basis": "sum_of_rounded_per_response_template_estimates",
          "notes": "Added from JY's HF PR21 at immutable revision 5d04d3b4b91dceb4c90e8592ce3235c1965c9db2; PR open at inventory. All 50 final tasks completed. No-preload track and original run contract retained. Cost is the sum of rounded per-response template estimates from final-task logs, not an invoice. Published-rate reconstruction independently agrees within the saved rounding bound. One prior infrastructure attempt is retained separately; the publisher's $16.43 total includes it, while this final-task row excludes its cost and usage.",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "5cde42fbcc216b69d434936e38fbf01c56a5e77904cd57d01aaa963fb7dbeea7",
          "run_terminal_events": "[\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 0.3190882,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "sum_of_rounded_per_response_template_estimates. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "HF PR21 publisher's repository-configured rate snapshot checked 2026-09-15; uncached input/cached input/output=5/0.50/30 USD per million; not independently reconciled with billing; cache writes charged within uncached input by this formula",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/discussions/21",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-5.6-sol",
            "notes": "OpenAI's July 9, 2026 changelog announces public availability of GPT-5.6 Sol and Luna. The legacy June 26 date is not the public release date. Proprietary hosted model.",
            "model": "GPT-5.6 Sol",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_62_9b9e92"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/5d04d3b4b91dceb4c90e8592ce3235c1965c9db2/jykoh-gpt56sol-osworld-energy50-representative-modal-low-500steps-62.zip"
          ],
          "task_count": 50,
          "steps_per_task": 11.2,
          "tool_calls_per_task": 45.34,
          "model_responses_per_task": 12.16,
          "variant": "",
          "series": "GPT-5.6 Sol",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "gpt54",
            "benchmark_content_hash": "fc402fbaf9f4a14426ee72a7af3ca8b8565b48cf92d52ac53ad19f5df0772876",
            "track_name": "api-cpu-agents1-no-preload",
            "measurement_hash": "47f89119bb0e8086be3679fd64d0124328e04c78154957992dbb350d2dafe18c",
            "eval_algorithm": "shared-agent-no-preload@1",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "default",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 47,
          "entry_name": "Meta Muse Spark 1.2 · low · svc_48_4e20fb",
          "model": "Meta Muse Spark 1.2",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.6361904360085809,
          "time_per_task_sec": 363.58951837,
          "average_time_per_task_sec": 363.58951837,
          "median_time_per_task_sec": 259.289296104,
          "cost_usd": 0.105404428,
          "turns_per_task": 23.94,
          "release_date": "2026-08-05",
          "average_output_tokens": 197.3015873015873,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_48_4e20fb",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark12-energy50-low-48.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "62ae3b1c46207e468b37b7d3616a009a4718b440f1c456b11cf31b63f3c3a8d0",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "reference_estimate",
          "cost_label": "API estimate · no cache",
          "cost_note": "API-equivalent estimate using recorded tokens and no input caching; not historical billing. Current OpenRouter public API-equivalent token price interval, not historical billing; lower assumes all input cached, upper none; excludes other charges",
          "cost_range_min_usd": 0.0030338749600000003,
          "cost_range_max_usd": 0.105404428,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://openrouter.ai/api/v1/models (snapshot 2026-09-13T17:46:52Z)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.2-contributor",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-08-05",
            "release_source": "https://research.meta.ai/blog/introducing-muse-code-and-muse-spark-1-2",
            "access_source": "https://research.meta.ai/blog/introducing-muse-code-and-muse-spark-1-2",
            "notes": "Proprietary model available through Muse Code and Meta Model API.",
            "model": "Meta Muse Spark 1.2",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_48_4e20fb"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark12-energy50-low-48.zip"
          ],
          "task_count": 50,
          "steps_per_task": 31.12,
          "tool_calls_per_task": 137.82,
          "model_responses_per_task": 23.94,
          "variant": "",
          "series": "Meta Muse Spark 1.2",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "meta",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 48,
          "entry_name": "Meta Muse Spark 1.2 · medium · svc_41_6ef6f1",
          "model": "Meta Muse Spark 1.2",
          "effort": "medium",
          "effort_rank": 2,
          "model_type": "closed",
          "performance": 0.6361904360085809,
          "time_per_task_sec": 466.21810665412,
          "average_time_per_task_sec": 466.21810665412,
          "median_time_per_task_sec": 358.864299247,
          "cost_usd": 0.136495792,
          "turns_per_task": 28.34,
          "release_date": "2026-08-05",
          "average_output_tokens": 274.8285109386027,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_41_6ef6f1",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark12-energy50-medium-41.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "8616b67de2f5ccc1225949fb2f4d6244e8dea03433e96d66237bd8118f6ec70f",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "reference_estimate",
          "cost_label": "API estimate · no cache",
          "cost_note": "API-equivalent estimate using recorded tokens and no input caching; not historical billing. Current OpenRouter public API-equivalent token price interval, not historical billing; lower assumes all input cached, upper none; excludes other charges",
          "cost_range_min_usd": 0.00425648928,
          "cost_range_max_usd": 0.136495792,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://openrouter.ai/api/v1/models (snapshot 2026-09-13T17:46:52Z)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.2-contributor",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-08-05",
            "release_source": "https://research.meta.ai/blog/introducing-muse-code-and-muse-spark-1-2",
            "access_source": "https://research.meta.ai/blog/introducing-muse-code-and-muse-spark-1-2",
            "notes": "Proprietary model available through Muse Code and Meta Model API.",
            "model": "Meta Muse Spark 1.2",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_41_6ef6f1"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark12-energy50-medium-41.zip"
          ],
          "task_count": 50,
          "steps_per_task": 44.2,
          "tool_calls_per_task": 152.84,
          "model_responses_per_task": 28.34,
          "variant": "",
          "series": "Meta Muse Spark 1.2",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "meta",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 126,
          "entry_name": "GPT-5.6 Luna · low · svc_63_ccb407",
          "model": "GPT-5.6 Luna",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.6181298513479518,
          "time_per_task_sec": 73.63877297137999,
          "average_time_per_task_sec": 73.63877297137999,
          "median_time_per_task_sec": 68.781423591,
          "cost_usd": 0.010669900000000001,
          "turns_per_task": 11.06,
          "release_date": "2026-07-09",
          "average_output_tokens": 96.08318264014467,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_63_ccb407",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/5d04d3b4b91dceb4c90e8592ce3235c1965c9db2/jykoh-gpt56luna-osworld-energy50-representative-modal-low-500steps-63.zip",
          "cost_basis": "sum_of_rounded_per_response_template_estimates",
          "notes": "Added from JY's HF PR21 at immutable revision 5d04d3b4b91dceb4c90e8592ce3235c1965c9db2; PR open at inventory. All 50 final tasks completed. No-preload track and original run contract retained. Cost is the sum of rounded per-response template estimates from final-task logs, not an invoice. Published-rate reconstruction independently agrees within the saved rounding bound.",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "402341ff59468534ed1e8c288c2713d6e700f5f626cc74a0f50fe9f203a2ba16",
          "run_terminal_events": "[\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 0.010669900000000001,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "sum_of_rounded_per_response_template_estimates. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "HF PR21 publisher's repository-configured rate snapshot checked 2026-09-15; uncached input/cached input/output=0.20/0.02/1.20 USD per million; not independently reconciled with billing; cache writes charged within uncached input by this formula",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/discussions/21",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
            "notes": "OpenAI's July 9, 2026 changelog announces public availability of GPT-5.6 Sol and Luna. The legacy June 26 date is not the public release date. Proprietary hosted model.",
            "model": "GPT-5.6 Luna",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_63_ccb407"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/5d04d3b4b91dceb4c90e8592ce3235c1965c9db2/jykoh-gpt56luna-osworld-energy50-representative-modal-low-500steps-63.zip"
          ],
          "task_count": 50,
          "steps_per_task": 10.12,
          "tool_calls_per_task": 37.66,
          "model_responses_per_task": 11.06,
          "variant": "Direct API",
          "series": "GPT-5.6 Luna · Direct API",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "gpt54",
            "benchmark_content_hash": "fc402fbaf9f4a14426ee72a7af3ca8b8565b48cf92d52ac53ad19f5df0772876",
            "track_name": "api-cpu-agents1-no-preload",
            "measurement_hash": "47f89119bb0e8086be3679fd64d0124328e04c78154957992dbb350d2dafe18c",
            "eval_algorithm": "shared-agent-no-preload@1",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "default",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 54,
          "entry_name": "Meta Muse Spark 1.3 · minimal · svc_67_b7d034",
          "model": "Meta Muse Spark 1.3",
          "effort": "minimal",
          "effort_rank": 0,
          "model_type": "closed",
          "performance": 0.6161904360085809,
          "time_per_task_sec": 264.43115914798,
          "average_time_per_task_sec": 264.43115914798,
          "median_time_per_task_sec": 150.7185996045,
          "cost_usd": 0.07891023,
          "turns_per_task": 19.76,
          "release_date": "2026-09-02",
          "average_output_tokens": 191.88765182186233,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_67_b7d034",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-energy50-minimal-67.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "6b1403871ca4fb7f55eec0bd533ab9edbfa57ecac5ce8a5ef8b4f4a7e7df2db2",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "reference_estimate",
          "cost_label": "API estimate · no cache",
          "cost_note": "API-equivalent estimate using recorded tokens and no input caching; not historical billing. Current OpenRouter public API-equivalent token price interval, not historical billing; lower assumes all input cached, upper none; excludes other charges",
          "cost_range_min_usd": 0.0023213778,
          "cost_range_max_usd": 0.07891023,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://openrouter.ai/api/v1/models (snapshot 2026-09-13T17:46:52Z)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.3-contributor",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-02",
            "release_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "access_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "notes": "Proprietary model available through Muse Code and Meta Model API.",
            "model": "Meta Muse Spark 1.3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_67_b7d034"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-energy50-minimal-67.zip"
          ],
          "task_count": 50,
          "steps_per_task": 32.04,
          "tool_calls_per_task": 59.52,
          "model_responses_per_task": 19.76,
          "variant": "",
          "series": "Meta Muse Spark 1.3",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "meta",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 46,
          "entry_name": "Meta Muse Spark 1.2 · high · svc_40_c6a678",
          "model": "Meta Muse Spark 1.2",
          "effort": "high",
          "effort_rank": 3,
          "model_type": "closed",
          "performance": 0.6161904360085809,
          "time_per_task_sec": 669.7751867468801,
          "average_time_per_task_sec": 669.7751867468801,
          "median_time_per_task_sec": 415.915365305,
          "cost_usd": 0.173570328,
          "turns_per_task": 33.74,
          "release_date": "2026-08-05",
          "average_output_tokens": 283.07409602845286,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_40_c6a678",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark12-energy50-high-40.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "c9666059bc62ca9423cda9d2c4bf54713618bdda73105e2fca09a3d03f6d8bc2",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "reference_estimate",
          "cost_label": "API estimate · no cache",
          "cost_note": "API-equivalent estimate using recorded tokens and no input caching; not historical billing. Current OpenRouter public API-equivalent token price interval, not historical billing; lower assumes all input cached, upper none; excludes other charges",
          "cost_range_min_usd": 0.00534338688,
          "cost_range_max_usd": 0.173570328,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://openrouter.ai/api/v1/models (snapshot 2026-09-13T17:46:52Z)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.2-contributor",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-08-05",
            "release_source": "https://research.meta.ai/blog/introducing-muse-code-and-muse-spark-1-2",
            "access_source": "https://research.meta.ai/blog/introducing-muse-code-and-muse-spark-1-2",
            "notes": "Proprietary model available through Muse Code and Meta Model API.",
            "model": "Meta Muse Spark 1.2",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_40_c6a678"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark12-energy50-high-40.zip"
          ],
          "task_count": 50,
          "steps_per_task": 45.72,
          "tool_calls_per_task": 145.58,
          "model_responses_per_task": 33.74,
          "variant": "",
          "series": "Meta Muse Spark 1.2",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "meta",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 22,
          "entry_name": "Gemini 3 Flash Preview · high · svc_81_c8b73b",
          "model": "Gemini 3 Flash Preview",
          "effort": "high",
          "effort_rank": 3,
          "model_type": "closed",
          "performance": 0.5961904360085809,
          "time_per_task_sec": 337.1563221826,
          "average_time_per_task_sec": 337.1563221826,
          "median_time_per_task_sec": 182.56473432099995,
          "cost_usd": 0.809073,
          "turns_per_task": 37.86,
          "release_date": "2025-12-17",
          "average_output_tokens": 136.46539883782356,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_81_c8b73b",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gemini3flash-energy50-high-modal-v1-81.zip",
          "cost_basis": "sum_of_rounded_per_response_template_estimates",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "c168ed65f4763877f28bd35f2c619e9b4c414af2b43298c8c1603609bf6c0379",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.809073,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "sum_of_rounded_per_response_template_estimates. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "sum_of_rounded_per_response_template_estimates",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gemini3flash-energy50-high-modal-v1-81.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2025-12-17",
            "release_source": "https://ai.google.dev/gemini-api/docs/deprecations",
            "access_source": "https://ai.google.dev/gemini-api/docs/models",
            "notes": "gemini-3-flash-preview; the legacy aggregate shortens the same model name to Gemini 3 Flash. Proprietary Gemini API model, not the open-weight Gemma family.",
            "model": "Gemini 3 Flash Preview",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_81_c8b73b"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gemini3flash-energy50-high-modal-v1-81.zip"
          ],
          "task_count": 50,
          "steps_per_task": 33.28,
          "tool_calls_per_task": 34.56,
          "model_responses_per_task": 37.86,
          "variant": "",
          "series": "Gemini 3 Flash Preview",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "gemini3_flash_preview",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 121,
          "entry_name": "GLM-5V Turbo · thinking_off · svc_87_ed8093",
          "model": "GLM-5V Turbo",
          "effort": "thinking_off",
          "effort_rank": 0,
          "model_type": "closed",
          "performance": 0.5780605846606292,
          "time_per_task_sec": 752.37900919524,
          "average_time_per_task_sec": 752.37900919524,
          "median_time_per_task_sec": 403.8254802854999,
          "cost_usd": 0.22032346399999997,
          "turns_per_task": 17.54,
          "release_date": "2026-04-01",
          "average_output_tokens": 464.6385404789054,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_87_ed8093",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61afda292c109e49aaa87f9dfc35f4e8bc5363a9/ljang-glm5v-energy50-thinking-off-pillow-fixed-87.zip",
          "cost_basis": "unavailable",
          "notes": "Publisher-designated corrected thinking-off/Pillow-fixed rerun; Pillow 11.3.0 installation is present in init.log. Recorded dollar cost and reasoning/cache token breakdown are unavailable, not zero. Separate reference-cost bounds use the cited OpenRouter rate snapshot and assume all versus no input cached; they are not billing.",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "e9a120b15db28697f225221551abd54170e3a6eb58df65baf231ffb9d3459ef0",
          "run_terminal_events": "[\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "reference_estimate",
          "cost_label": "API estimate · no cache",
          "cost_note": "API-equivalent estimate using recorded tokens and no input caching; not historical billing. Current OpenRouter public API-equivalent token price interval, not historical billing; lower assumes all input cached, upper none; excludes other charges",
          "cost_range_min_usd": 0.0701439248,
          "cost_range_max_usd": 0.22032346399999997,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://openrouter.ai/api/v1/models (snapshot 2026-09-15T20:56:37.449456+00:00)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "z-ai/glm-5v-turbo",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-04-01",
            "release_source": "https://docs.z.ai/release-notes/new-released",
            "access_source": "https://docs.z.ai/guides/vlm/glm-5v-turbo",
            "notes": "The publisher's release notes date this API model April 1; the GLM-V repository posted its announcement April 2. GLM-5V-Turbo is not one of the downloadable GLM-V models; the GLM family name does not establish open weights.",
            "model": "GLM-5V Turbo",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_87_ed8093"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61afda292c109e49aaa87f9dfc35f4e8bc5363a9/ljang-glm5v-energy50-thinking-off-pillow-fixed-87.zip"
          ],
          "task_count": 50,
          "steps_per_task": 16.86,
          "tool_calls_per_task": 21.52,
          "model_responses_per_task": 17.54,
          "variant": "",
          "series": "GLM-5V Turbo",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "glm5v_turbo",
            "benchmark_content_hash": "fc402fbaf9f4a14426ee72a7af3ca8b8565b48cf92d52ac53ad19f5df0772876",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "34761619b59911dddf1748656f82a27a24896a268da28e7192a6407cad0f67b7",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 24,
          "entry_name": "Gemini 3 Flash Preview · medium · svc_83_d83faa",
          "model": "Gemini 3 Flash Preview",
          "effort": "medium",
          "effort_rank": 2,
          "model_type": "closed",
          "performance": 0.5761904360085809,
          "time_per_task_sec": 268.0254726477,
          "average_time_per_task_sec": 268.0254726477,
          "median_time_per_task_sec": 151.91498372450002,
          "cost_usd": 0.592287,
          "turns_per_task": 32.52,
          "release_date": "2025-12-17",
          "average_output_tokens": 126.83025830258303,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_83_d83faa",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gemini3flash-energy50-medium-modal-v1-83.zip",
          "cost_basis": "sum_of_rounded_per_response_template_estimates",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "6dc6a3bdccd82dead0e1cb4ec5159bad1dc74450b53319a4536473327a193cc8",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 0.592287,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "sum_of_rounded_per_response_template_estimates. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "sum_of_rounded_per_response_template_estimates",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gemini3flash-energy50-medium-modal-v1-83.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2025-12-17",
            "release_source": "https://ai.google.dev/gemini-api/docs/deprecations",
            "access_source": "https://ai.google.dev/gemini-api/docs/models",
            "notes": "gemini-3-flash-preview; the legacy aggregate shortens the same model name to Gemini 3 Flash. Proprietary Gemini API model, not the open-weight Gemma family.",
            "model": "Gemini 3 Flash Preview",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_83_d83faa"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gemini3flash-energy50-medium-modal-v1-83.zip"
          ],
          "task_count": 50,
          "steps_per_task": 28.26,
          "tool_calls_per_task": 30.68,
          "model_responses_per_task": 32.52,
          "variant": "",
          "series": "Gemini 3 Flash Preview",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "gemini3_flash_preview",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 14,
          "entry_name": "GLM-5V Turbo · thinking_on · svc_49_c74b13",
          "model": "GLM-5V Turbo",
          "effort": "thinking_on",
          "effort_rank": 6,
          "model_type": "closed",
          "performance": 0.5380605846606291,
          "time_per_task_sec": 359.24417814123996,
          "average_time_per_task_sec": 359.24417814123996,
          "median_time_per_task_sec": 195.8833519480001,
          "cost_usd": 0.23496064,
          "turns_per_task": 18.62,
          "release_date": "2026-04-01",
          "average_output_tokens": 459.8721804511278,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_49_c74b13",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-glm5v-energy50-thinking-on-49.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "84a591295e7941a18bcf97e48306ff12cd443f7cdf86a4e981da464044482d63",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "reference_estimate",
          "cost_label": "API estimate · no cache",
          "cost_note": "API-equivalent estimate using recorded tokens and no input caching; not historical billing. Current OpenRouter public API-equivalent token price interval, not historical billing; lower assumes all input cached, upper none; excludes other charges",
          "cost_range_min_usd": 0.07439315199999999,
          "cost_range_max_usd": 0.23496064,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://openrouter.ai/api/v1/models (snapshot 2026-09-13T17:46:52Z)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "z-ai/glm-5v-turbo",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-04-01",
            "release_source": "https://docs.z.ai/release-notes/new-released",
            "access_source": "https://docs.z.ai/guides/vlm/glm-5v-turbo",
            "notes": "The publisher's release notes date this API model April 1; the GLM-V repository posted its announcement April 2. GLM-5V-Turbo is not one of the downloadable GLM-V models; the GLM family name does not establish open weights.",
            "model": "GLM-5V Turbo",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_49_c74b13"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-glm5v-energy50-thinking-on-49.zip"
          ],
          "task_count": 50,
          "steps_per_task": 17.86,
          "tool_calls_per_task": 19.24,
          "model_responses_per_task": 18.62,
          "variant": "",
          "series": "GLM-5V Turbo",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "glm5v_turbo",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 49,
          "entry_name": "Meta Muse Spark 1.2 · minimal · svc_47_1eed89",
          "model": "Meta Muse Spark 1.2",
          "effort": "minimal",
          "effort_rank": 0,
          "model_type": "closed",
          "performance": 0.536190436008581,
          "time_per_task_sec": 248.79153390881999,
          "average_time_per_task_sec": 248.79153390881999,
          "median_time_per_task_sec": 186.83313444700002,
          "cost_usd": 0.069708796,
          "turns_per_task": 18.72,
          "release_date": "2026-08-05",
          "average_output_tokens": 127.08333333333334,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_47_1eed89",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark12-energy50-minimal-47.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "15c824ae416e3cc959757170f868727f8eea232fc664e57b17e57e497cd9f6f0",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "reference_estimate",
          "cost_label": "API estimate · no cache",
          "cost_note": "API-equivalent estimate using recorded tokens and no input caching; not historical billing. Current OpenRouter public API-equivalent token price interval, not historical billing; lower assumes all input cached, upper none; excludes other charges",
          "cost_range_min_usd": 0.0018604599199999999,
          "cost_range_max_usd": 0.069708796,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "https://openrouter.ai/api/v1/models (snapshot 2026-09-13T17:46:52Z)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.2-contributor",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-08-05",
            "release_source": "https://research.meta.ai/blog/introducing-muse-code-and-muse-spark-1-2",
            "access_source": "https://research.meta.ai/blog/introducing-muse-code-and-muse-spark-1-2",
            "notes": "Proprietary model available through Muse Code and Meta Model API.",
            "model": "Meta Muse Spark 1.2",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_47_1eed89"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark12-energy50-minimal-47.zip"
          ],
          "task_count": 50,
          "steps_per_task": 23.2,
          "tool_calls_per_task": 87.38,
          "model_responses_per_task": 18.72,
          "variant": "",
          "series": "Meta Muse Spark 1.2",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "meta",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 23,
          "entry_name": "Gemini 3 Flash Preview · low · svc_88_da1388",
          "model": "Gemini 3 Flash Preview",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.336190436008581,
          "time_per_task_sec": 492.15841232933997,
          "average_time_per_task_sec": 492.15841232933997,
          "median_time_per_task_sec": 489.7002096819999,
          "cost_usd": 1.4006396,
          "turns_per_task": 56.02,
          "release_date": "2025-12-17",
          "average_output_tokens": 28.533380935380222,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_88_da1388",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gemini3flash-energy50-low-modal-v1-88.zip",
          "cost_basis": "sum_of_rounded_per_response_template_estimates",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "45ef8b98246f640c3170da73d37b8428189c8dd6dc6f79be16089d7d6334cdd9",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 1.4006396,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "sum_of_rounded_per_response_template_estimates. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 50,
          "cost_total_tasks": 50,
          "cost_source": "sum_of_rounded_per_response_template_estimates",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gemini3flash-energy50-low-modal-v1-88.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2025-12-17",
            "release_source": "https://ai.google.dev/gemini-api/docs/deprecations",
            "access_source": "https://ai.google.dev/gemini-api/docs/models",
            "notes": "gemini-3-flash-preview; the legacy aggregate shortens the same model name to Gemini 3 Flash. Proprietary Gemini API model, not the open-weight Gemma family.",
            "model": "Gemini 3 Flash Preview",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_88_da1388"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gemini3flash-energy50-low-modal-v1-88.zip"
          ],
          "task_count": 50,
          "steps_per_task": 39.52,
          "tool_calls_per_task": 40.26,
          "model_responses_per_task": 56.02,
          "variant": "",
          "series": "Gemini 3 Flash Preview",
          "source_metadata": {
            "split_dset_name": "osworld-energy50-representative",
            "benchmark_name": "osworld-energy50-representative",
            "benchmark_version": "0.1",
            "template_name": "gemini3_flash_preview",
            "benchmark_content_hash": "92b1dd8a2e0dd9776c700def058720fd0b9980bac1c7c5fa9e2f5817f513a416",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "92fc0b3011dc577b479c6e648f7dc7d8081d977dda6937693bbbafec7e81c13d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        }
      ],
      "excluded_evaluations": [
        {
          "model": "GLM-5V Turbo",
          "effort": "thinking_on",
          "run_id": "svc_50_592ed5",
          "reason": "superseded configuration label",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-glm5v-energy50-thinking-off-50.zip",
          "variant": "",
          "series": "GLM-5V Turbo",
          "model_type": "closed",
          "release_date": "2026-04-01",
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-04-01",
            "release_source": "https://docs.z.ai/release-notes/new-released",
            "access_source": "https://docs.z.ai/guides/vlm/glm-5v-turbo",
            "notes": "The publisher's release notes date this API model April 1; the GLM-V repository posted its announcement April 2. GLM-5V-Turbo is not one of the downloadable GLM-V models; the GLM family name does not establish open weights.",
            "model": "GLM-5V Turbo",
            "verified_on": "2026-09-16"
          }
        },
        {
          "model": "Gemini 3.8 Flash",
          "effort": "medium",
          "run_id": "svc_97_82e253",
          "reason": "Older reference (2026-09-03); retaining newer medium run svc_61_1ec4be (2026-09-13, no preload). Not a failed run.",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gemini38flash-energy50-medium-modal-v1-97.zip",
          "variant": "",
          "series": "Gemini 3.8 Flash",
          "model_type": "closed",
          "release_date": "2026-09-02",
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-02",
            "release_source": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/3-8-flash-and-3-8-flash-cyber/",
            "access_source": "https://ai.google.dev/gemini-api/docs/models",
            "notes": "Proprietary Gemini API model.",
            "model": "Gemini 3.8 Flash",
            "verified_on": "2026-09-16"
          }
        },
        {
          "model": "Meta Muse Spark 1.1",
          "effort": "high",
          "run_id": "svc_29_8c9590",
          "reason": "Rate-limited playground contrast; contributor-tier high run svc_58_38520a retained.",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/46b7e02ccd5eb1c3422b5ee6c60f40d84d19ec8e/ljang-musespark11-playground-energy50-high-29.zip",
          "variant": "",
          "series": "Meta Muse Spark 1.1",
          "model_type": "closed",
          "release_date": "2026-07-09",
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "access_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "notes": "Proprietary Muse Spark model served through Meta Model API; not an open-weight Llama or Muse Glimmer release.",
            "model": "Meta Muse Spark 1.1",
            "verified_on": "2026-09-16"
          }
        },
        {
          "model": "Meta Muse Spark 1.1",
          "effort": "low",
          "run_id": "svc_31_c00240",
          "reason": "Rate-limited playground contrast; contributor-tier low run svc_60_72500b retained.",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/46b7e02ccd5eb1c3422b5ee6c60f40d84d19ec8e/ljang-musespark11-playground-energy50-low-31.zip",
          "variant": "",
          "series": "Meta Muse Spark 1.1",
          "model_type": "closed",
          "release_date": "2026-07-09",
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "access_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "notes": "Proprietary Muse Spark model served through Meta Model API; not an open-weight Llama or Muse Glimmer release.",
            "model": "Meta Muse Spark 1.1",
            "verified_on": "2026-09-16"
          }
        },
        {
          "model": "Meta Muse Spark 1.1",
          "effort": "medium",
          "run_id": "svc_30_d46514",
          "reason": "Rate-limited playground contrast; contributor-tier medium run svc_59_d8a6d4 retained.",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/46b7e02ccd5eb1c3422b5ee6c60f40d84d19ec8e/ljang-musespark11-playground-energy50-medium-30.zip",
          "variant": "",
          "series": "Meta Muse Spark 1.1",
          "model_type": "closed",
          "release_date": "2026-07-09",
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "access_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "notes": "Proprietary Muse Spark model served through Meta Model API; not an open-weight Llama or Muse Glimmer release.",
            "model": "Meta Muse Spark 1.1",
            "verified_on": "2026-09-16"
          }
        },
        {
          "model": "Meta Muse Spark 1.1",
          "effort": "minimal",
          "run_id": "svc_34_8c4108",
          "reason": "Rate-limited playground contrast; contributor-tier minimal run svc_61_96ce3f retained.",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/46b7e02ccd5eb1c3422b5ee6c60f40d84d19ec8e/ljang-musespark11-playground-energy50-minimal-34.zip",
          "variant": "",
          "series": "Meta Muse Spark 1.1",
          "model_type": "closed",
          "release_date": "2026-07-09",
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "access_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "notes": "Proprietary Muse Spark model served through Meta Model API; not an open-weight Llama or Muse Glimmer release.",
            "model": "Meta Muse Spark 1.1",
            "verified_on": "2026-09-16"
          }
        },
        {
          "model": "Meta Muse Spark 1.1",
          "effort": "xhigh",
          "run_id": "svc_53_1cade5",
          "reason": "billing-affected",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/46b7e02ccd5eb1c3422b5ee6c60f40d84d19ec8e/ljang-musespark11-energy50-xhigh-53.zip",
          "variant": "",
          "series": "Meta Muse Spark 1.1",
          "model_type": "closed",
          "release_date": "2026-07-09",
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "access_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "notes": "Proprietary Muse Spark model served through Meta Model API; not an open-weight Llama or Muse Glimmer release.",
            "model": "Meta Muse Spark 1.1",
            "verified_on": "2026-09-16"
          }
        },
        {
          "model": "Meta Muse Spark 1.1",
          "effort": "xhigh",
          "run_id": "svc_32_109178",
          "reason": "Rate-limited playground contrast; contributor-tier xhigh run svc_69_41f88a retained.",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/46b7e02ccd5eb1c3422b5ee6c60f40d84d19ec8e/ljang-musespark11-playground-energy50-xhigh-32.zip",
          "variant": "",
          "series": "Meta Muse Spark 1.1",
          "model_type": "closed",
          "release_date": "2026-07-09",
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-09",
            "release_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "access_source": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
            "notes": "Proprietary Muse Spark model served through Meta Model API; not an open-weight Llama or Muse Glimmer release.",
            "model": "Meta Muse Spark 1.1",
            "verified_on": "2026-09-16"
          }
        }
      ],
      "supporting_rows": 19,
      "task_counts": [
        50
      ],
      "paper_subset": {
        "label": "OSWorld",
        "selected_tasks": 50,
        "eligible_tasks": 295
      }
    },
    "osworld2-k52": {
      "name": "osworld2-k52",
      "label": "OSWorld 2.0 · 52 tasks",
      "href": "dataset-osworld2-k52.html",
      "records": [
        {
          "entry_id": 94,
          "entry_name": "GPT-6 Astra · xhigh · svc_119_69f289",
          "model": "GPT-6 Astra",
          "effort": "xhigh",
          "effort_rank": 4,
          "model_type": "closed",
          "performance": 0.769016160783479,
          "time_per_task_sec": 1237.2831666631346,
          "average_time_per_task_sec": 1237.2831666631346,
          "median_time_per_task_sec": 970.0857832639999,
          "cost_usd": 8.174599923076924,
          "turns_per_task": null,
          "release_date": "2026-09-03",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_119_69f289",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gpt6astra-osworld2-k52-xhigh-modal-v1-119.zip",
          "cost_basis": "Astra_standard_short_context_API_equivalent_not_subscription_bill",
          "notes": "Cost is short-context API-equivalent usage, not a subscription payment; full tool/response trace unavailable in CLI export.",
          "quality_flags": "{\"Astra_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":52,\"codex_model_response_count_and_full_tool_calls_unavailable\":52}",
          "selected_task_seed_set_sha256": "023db06c919361441f78268d26d2d2d7c995c3ac384cf0e878911224d652c73c",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 8.174599923076924,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Astra_standard_short_context_API_equivalent_not_subscription_bill. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 52,
          "cost_total_tasks": 52,
          "cost_source": "HF COST-pranjal-eval137.json pinned pricing snapshot; input/cache/write/output=10/1/12.5/50 USD per million",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9b541e31ddb6c7ffcfde1f460f5c2d449f625072/COST-pranjal-eval137.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-03",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-6-astra",
            "notes": "OpenAI's September 3, 2026 changelog announces GPT-6 Astra. Proprietary hosted model.",
            "model": "GPT-6 Astra",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_119_69f289"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gpt6astra-osworld2-k52-xhigh-modal-v1-119.zip"
          ],
          "task_count": 52,
          "steps_per_task": 52.80769230769231,
          "tool_calls_per_task": 194.09615384615384,
          "model_responses_per_task": null,
          "variant": "",
          "series": "GPT-6 Astra",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "codex_cli",
            "benchmark_content_hash": "b8075161613a4fdc7823a469d5e66237dec8335e34fd24f4e29f5c0b79c68d6e",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "0a1e9a0b6132d82a4ebefed299f1bc30ed8de7436d8df4ee793ce90d8719791a",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 91,
          "entry_name": "GPT-6 Astra · high · svc_73_9d68f5",
          "model": "GPT-6 Astra",
          "effort": "high",
          "effort_rank": 3,
          "model_type": "closed",
          "performance": 0.750164311586559,
          "time_per_task_sec": 914.0425841820191,
          "average_time_per_task_sec": 914.0425841820191,
          "median_time_per_task_sec": 824.5325481560001,
          "cost_usd": 7.390932730769231,
          "turns_per_task": null,
          "release_date": "2026-09-03",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_73_9d68f5",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-codex-astra-high-osworld2-k52-73.zip",
          "cost_basis": "Astra_standard_short_context_API_equivalent_not_subscription_bill",
          "notes": "Cost is short-context API-equivalent usage, not a subscription payment; full tool/response trace unavailable in CLI export.",
          "quality_flags": "{\"Astra_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":52,\"codex_model_response_count_and_full_tool_calls_unavailable\":52}",
          "selected_task_seed_set_sha256": "44c85c2f6a55d764ef1e25c706d3c1e2dab10f6f31d5796c1cd85341c76b2cff",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 7.390932730769231,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Astra_standard_short_context_API_equivalent_not_subscription_bill. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 52,
          "cost_total_tasks": 52,
          "cost_source": "HF COST-pranjal-eval137.json pinned pricing snapshot; input/cache/write/output=10/1/12.5/50 USD per million",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9b541e31ddb6c7ffcfde1f460f5c2d449f625072/COST-pranjal-eval137.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-03",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-6-astra",
            "notes": "OpenAI's September 3, 2026 changelog announces GPT-6 Astra. Proprietary hosted model.",
            "model": "GPT-6 Astra",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_73_9d68f5"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-codex-astra-high-osworld2-k52-73.zip"
          ],
          "task_count": 52,
          "steps_per_task": 51.28846153846154,
          "tool_calls_per_task": 149.01923076923077,
          "model_responses_per_task": null,
          "variant": "",
          "series": "GPT-6 Astra",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "codex_cli",
            "benchmark_content_hash": "b8075161613a4fdc7823a469d5e66237dec8335e34fd24f4e29f5c0b79c68d6e",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "0a1e9a0b6132d82a4ebefed299f1bc30ed8de7436d8df4ee793ce90d8719791a",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 92,
          "entry_name": "GPT-6 Astra · low · svc_123_d9dc75",
          "model": "GPT-6 Astra",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.6817110435279882,
          "time_per_task_sec": 904.2353977893654,
          "average_time_per_task_sec": 904.2353977893654,
          "median_time_per_task_sec": 677.1419179625,
          "cost_usd": 6.336339961538462,
          "turns_per_task": null,
          "release_date": "2026-09-03",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_123_d9dc75",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gpt6astra-osworld2-k52-low-modal-v1-123.zip",
          "cost_basis": "Astra_standard_short_context_API_equivalent_not_subscription_bill",
          "notes": "Cost is short-context API-equivalent usage, not a subscription payment; full tool/response trace unavailable in CLI export.",
          "quality_flags": "{\"Astra_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":52,\"codex_model_response_count_and_full_tool_calls_unavailable\":52}",
          "selected_task_seed_set_sha256": "d5629c64cd3611b1369f96f5be611214c1d154d83bd2630c7a2457711cb0fe5e",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 6.336339961538462,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Astra_standard_short_context_API_equivalent_not_subscription_bill. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 52,
          "cost_total_tasks": 52,
          "cost_source": "HF COST-pranjal-eval137.json pinned pricing snapshot; input/cache/write/output=10/1/12.5/50 USD per million",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9b541e31ddb6c7ffcfde1f460f5c2d449f625072/COST-pranjal-eval137.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-03",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-6-astra",
            "notes": "OpenAI's September 3, 2026 changelog announces GPT-6 Astra. Proprietary hosted model.",
            "model": "GPT-6 Astra",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_123_d9dc75"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/pranjal-gpt6astra-osworld2-k52-low-modal-v1-123.zip"
          ],
          "task_count": 52,
          "steps_per_task": 42.57692307692308,
          "tool_calls_per_task": 151.40384615384616,
          "model_responses_per_task": null,
          "variant": "",
          "series": "GPT-6 Astra",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "codex_cli",
            "benchmark_content_hash": "b8075161613a4fdc7823a469d5e66237dec8335e34fd24f4e29f5c0b79c68d6e",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "0a1e9a0b6132d82a4ebefed299f1bc30ed8de7436d8df4ee793ce90d8719791a",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 93,
          "entry_name": "GPT-6 Astra · medium · svc_74_19ec78",
          "model": "GPT-6 Astra",
          "effort": "medium",
          "effort_rank": 2,
          "model_type": "closed",
          "performance": 0.6718250285536133,
          "time_per_task_sec": 828.1029622646154,
          "average_time_per_task_sec": 828.1029622646154,
          "median_time_per_task_sec": 700.9397634865,
          "cost_usd": 6.519167346153846,
          "turns_per_task": null,
          "release_date": "2026-09-03",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_74_19ec78",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-codex-astra-medium-osworld2-k52-74.zip",
          "cost_basis": "Astra_standard_short_context_API_equivalent_not_subscription_bill",
          "notes": "Cost is short-context API-equivalent usage, not a subscription payment; full tool/response trace unavailable in CLI export.",
          "quality_flags": "{\"Astra_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":52,\"codex_model_response_count_and_full_tool_calls_unavailable\":52}",
          "selected_task_seed_set_sha256": "7df81f3277bfc9e1c1267d42827e7cc3d9b4b0b8e16f9ce53fc1b5ce166bb575",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 6.519167346153846,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Astra_standard_short_context_API_equivalent_not_subscription_bill. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 52,
          "cost_total_tasks": 52,
          "cost_source": "HF COST-pranjal-eval137.json pinned pricing snapshot; input/cache/write/output=10/1/12.5/50 USD per million",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9b541e31ddb6c7ffcfde1f460f5c2d449f625072/COST-pranjal-eval137.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-03",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-6-astra",
            "notes": "OpenAI's September 3, 2026 changelog announces GPT-6 Astra. Proprietary hosted model.",
            "model": "GPT-6 Astra",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_74_19ec78"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-codex-astra-medium-osworld2-k52-74.zip"
          ],
          "task_count": 52,
          "steps_per_task": 45.46153846153846,
          "tool_calls_per_task": 151.07692307692307,
          "model_responses_per_task": null,
          "variant": "",
          "series": "GPT-6 Astra",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "codex_cli",
            "benchmark_content_hash": "b8075161613a4fdc7823a469d5e66237dec8335e34fd24f4e29f5c0b79c68d6e",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "0a1e9a0b6132d82a4ebefed299f1bc30ed8de7436d8df4ee793ce90d8719791a",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 95,
          "entry_name": "Gemini 3.8 Flash · high · svc_55_ffdb2e",
          "model": "Gemini 3.8 Flash",
          "effort": "high",
          "effort_rank": 3,
          "model_type": "closed",
          "performance": 0.6017515083251204,
          "time_per_task_sec": 3058.7615230564998,
          "average_time_per_task_sec": 3058.7615230564998,
          "median_time_per_task_sec": 2591.0340092750002,
          "cost_usd": 5.019472442307693,
          "turns_per_task": 218.48076923076923,
          "release_date": "2026-09-02",
          "average_output_tokens": 330.2302614206496,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_55_ffdb2e",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gemini38flash-high-osworld2-k52-55.zip",
          "cost_basis": "recorded_template_price_estimate",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "d9a87cfe93d2f2be82ba7a8d444fe66be2bd0aa3e9853446960d3872c5a29052",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 5.019472442307693,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "recorded_template_price_estimate. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 52,
          "cost_total_tasks": 52,
          "cost_source": "recorded_template_price_estimate",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gemini38flash-high-osworld2-k52-55.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-02",
            "release_source": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/3-8-flash-and-3-8-flash-cyber/",
            "access_source": "https://ai.google.dev/gemini-api/docs/models",
            "notes": "Proprietary Gemini API model.",
            "model": "Gemini 3.8 Flash",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_55_ffdb2e"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gemini38flash-high-osworld2-k52-55.zip"
          ],
          "task_count": 52,
          "steps_per_task": 216.53846153846155,
          "tool_calls_per_task": 331.88461538461536,
          "model_responses_per_task": 218.48076923076923,
          "variant": "",
          "series": "Gemini 3.8 Flash",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "gemini_new",
            "benchmark_content_hash": "b8075161613a4fdc7823a469d5e66237dec8335e34fd24f4e29f5c0b79c68d6e",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "6ef9757c308296114d62334952b25f707a8ebc7a87aedfee663bf3e44f247d71",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 97,
          "entry_name": "Gemini 3.8 Flash · medium · svc_54_a54e88",
          "model": "Gemini 3.8 Flash",
          "effort": "medium",
          "effort_rank": 2,
          "model_type": "closed",
          "performance": 0.5993102689627893,
          "time_per_task_sec": 2051.8210410221154,
          "average_time_per_task_sec": 2051.8210410221154,
          "median_time_per_task_sec": 1559.9580825889998,
          "cost_usd": 2.9147833240384617,
          "turns_per_task": 169.65384615384616,
          "release_date": "2026-09-02",
          "average_output_tokens": 291.9569258671503,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_54_a54e88",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gemini38flash-medium-osworld2-k52-54.zip",
          "cost_basis": "recorded_template_price_estimate",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "10cdf5e275fa6b52d903057e7fa958b50d8bc698defd0aadf7a86619613181e6",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 2.9147833240384617,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "recorded_template_price_estimate. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 52,
          "cost_total_tasks": 52,
          "cost_source": "recorded_template_price_estimate",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gemini38flash-medium-osworld2-k52-54.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-02",
            "release_source": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/3-8-flash-and-3-8-flash-cyber/",
            "access_source": "https://ai.google.dev/gemini-api/docs/models",
            "notes": "Proprietary Gemini API model.",
            "model": "Gemini 3.8 Flash",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_54_a54e88"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gemini38flash-medium-osworld2-k52-54.zip"
          ],
          "task_count": 52,
          "steps_per_task": 167.78846153846155,
          "tool_calls_per_task": 252.65384615384616,
          "model_responses_per_task": 169.65384615384616,
          "variant": "",
          "series": "Gemini 3.8 Flash",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "gemini_new",
            "benchmark_content_hash": "b8075161613a4fdc7823a469d5e66237dec8335e34fd24f4e29f5c0b79c68d6e",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "6ef9757c308296114d62334952b25f707a8ebc7a87aedfee663bf3e44f247d71",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 88,
          "entry_name": "Claude Opus 5 · low · svc_57_addd34",
          "model": "Claude Opus 5",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.5206271724264461,
          "time_per_task_sec": 1385.680655156154,
          "average_time_per_task_sec": 1385.680655156154,
          "median_time_per_task_sec": 1120.803646337,
          "cost_usd": 10.006317384615384,
          "turns_per_task": 158.28846153846155,
          "release_date": "2026-07-24",
          "average_output_tokens": 194.70878386587293,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_57_addd34",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/a4d1d2377735560120d59b2138a61b7cce9286bc/jykoh-claude-opus-5-osworld2-k52-modal-low-500steps-57.zip",
          "cost_basis": "Claude_standard_global_list_price_estimate_from_logged_usage",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "cc04e79640e2505e847d307819c2a6f5747ac66801834bf57e29e80bab8a40f7",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 10.006317384615384,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Claude_standard_global_list_price_estimate_from_logged_usage. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 52,
          "cost_total_tasks": 52,
          "cost_source": "HF PR16 price snapshot checked 2026-09-13; Opus5 5/6.25/.5/25; Sonnet5 2/2.5/.2/10 USD per million input/5m-write/read/output",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/discussions/16",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-07-24",
            "release_source": "https://www.anthropic.com/news/claude-opus-5",
            "access_source": "https://www.anthropic.com/news/claude-opus-5",
            "notes": "Proprietary Claude model, available through Anthropic and hosted partners.",
            "model": "Claude Opus 5",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_57_addd34"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/a4d1d2377735560120d59b2138a61b7cce9286bc/jykoh-claude-opus-5-osworld2-k52-modal-low-500steps-57.zip"
          ],
          "task_count": 52,
          "steps_per_task": 146.51923076923077,
          "tool_calls_per_task": 316.1730769230769,
          "model_responses_per_task": 158.28846153846155,
          "variant": "",
          "series": "Claude Opus 5",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "claude",
            "benchmark_content_hash": "b8075161613a4fdc7823a469d5e66237dec8335e34fd24f4e29f5c0b79c68d6e",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "6ef9757c308296114d62334952b25f707a8ebc7a87aedfee663bf3e44f247d71",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 89,
          "entry_name": "Claude Sonnet 5 · low · svc_58_909a13",
          "model": "Claude Sonnet 5",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.39079439506141084,
          "time_per_task_sec": 1993.6084919779616,
          "average_time_per_task_sec": 1993.6084919779616,
          "median_time_per_task_sec": 1664.6346779239998,
          "cost_usd": 8.863014275,
          "turns_per_task": 253.5,
          "release_date": "2026-06-30",
          "average_output_tokens": 224.92429069943864,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_58_909a13",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/a4d1d2377735560120d59b2138a61b7cce9286bc/jykoh-claude-sonnet-5-osworld2-k52-modal-low-500steps-58.zip",
          "cost_basis": "Claude_standard_global_list_price_estimate_from_logged_usage",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "b3ebdcee71e9ab7bdc1ac0e52ee4e5a4721ed19dbe36eeaf406516b7455c9894",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 8.863014275,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Claude_standard_global_list_price_estimate_from_logged_usage. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 52,
          "cost_total_tasks": 52,
          "cost_source": "HF PR16 price snapshot checked 2026-09-13; Opus5 5/6.25/.5/25; Sonnet5 2/2.5/.2/10 USD per million input/5m-write/read/output",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/discussions/16",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-06-30",
            "release_source": "https://www.anthropic.com/news/claude-sonnet-5",
            "access_source": "https://www.anthropic.com/news/claude-sonnet-5",
            "notes": "Proprietary Claude model, available through Anthropic and hosted partners.",
            "model": "Claude Sonnet 5",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_58_909a13"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/a4d1d2377735560120d59b2138a61b7cce9286bc/jykoh-claude-sonnet-5-osworld2-k52-modal-low-500steps-58.zip"
          ],
          "task_count": 52,
          "steps_per_task": 237.25,
          "tool_calls_per_task": 378.53846153846155,
          "model_responses_per_task": 253.5,
          "variant": "",
          "series": "Claude Sonnet 5",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "claude",
            "benchmark_content_hash": "b8075161613a4fdc7823a469d5e66237dec8335e34fd24f4e29f5c0b79c68d6e",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "6ef9757c308296114d62334952b25f707a8ebc7a87aedfee663bf3e44f247d71",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 96,
          "entry_name": "Gemini 3.8 Flash · low · svc_56_08ebc5",
          "model": "Gemini 3.8 Flash",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.3529254812275505,
          "time_per_task_sec": 1386.6517991032115,
          "average_time_per_task_sec": 1386.6517991032115,
          "median_time_per_task_sec": 974.1764620825002,
          "cost_usd": 1.8751714038461533,
          "turns_per_task": 128.82692307692307,
          "release_date": "2026-09-02",
          "average_output_tokens": 154.00194058814748,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_56_08ebc5",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gemini38flash-low-osworld2-k52-56.zip",
          "cost_basis": "recorded_template_price_estimate",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "1da4d7aac96b5030c83abe8b2eb326ba07c388a0055f6cc6519b9bdaa6e1cddc",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 1.8751714038461533,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "recorded_template_price_estimate. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 52,
          "cost_total_tasks": 52,
          "cost_source": "recorded_template_price_estimate",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gemini38flash-low-osworld2-k52-56.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-02",
            "release_source": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/3-8-flash-and-3-8-flash-cyber/",
            "access_source": "https://ai.google.dev/gemini-api/docs/models",
            "notes": "Proprietary Gemini API model.",
            "model": "Gemini 3.8 Flash",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_56_08ebc5"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/jykoh-gemini38flash-low-osworld2-k52-56.zip"
          ],
          "task_count": 52,
          "steps_per_task": 127.03846153846153,
          "tool_calls_per_task": 200.75,
          "model_responses_per_task": 128.82692307692307,
          "variant": "",
          "series": "Gemini 3.8 Flash",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "gemini_new",
            "benchmark_content_hash": "b8075161613a4fdc7823a469d5e66237dec8335e34fd24f4e29f5c0b79c68d6e",
            "track_name": "open-l40s-shared-agents1",
            "measurement_hash": "6ef9757c308296114d62334952b25f707a8ebc7a87aedfee663bf3e44f247d71",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 104,
          "entry_name": "Meta Muse Spark 1.3 · minimal · svc_80_6fea36",
          "model": "Meta Muse Spark 1.3",
          "effort": "minimal",
          "effort_rank": 0,
          "model_type": "closed",
          "performance": 0.3396830047424765,
          "time_per_task_sec": 1515.9694856658846,
          "average_time_per_task_sec": 1515.9694856658846,
          "median_time_per_task_sec": 1518.2948088555,
          "cost_usd": 0.39032538076923073,
          "turns_per_task": 66.01923076923077,
          "release_date": "2026-09-02",
          "average_output_tokens": 244.6201572968249,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_80_6fea36",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-osworld2-k52-minimal-80.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "1cd64a928b967a5f7bd1ea7fd659738d3eac7003939ce3bc4449de586419b2d1",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "reference_estimate",
          "cost_label": "API estimate · no cache",
          "cost_note": "API-equivalent estimate using recorded tokens and no input caching; not historical billing. Current OpenRouter public API-equivalent token price interval, not historical billing; lower assumes all input cached, upper none; excludes other charges",
          "cost_range_min_usd": 0.010971836,
          "cost_range_max_usd": 0.39032538076923073,
          "cost_coverage_tasks": 52,
          "cost_total_tasks": 52,
          "cost_source": "https://openrouter.ai/api/v1/models (snapshot 2026-09-13T17:46:52Z)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.3-contributor",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-02",
            "release_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "access_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "notes": "Proprietary model available through Muse Code and Meta Model API.",
            "model": "Meta Muse Spark 1.3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_80_6fea36"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-osworld2-k52-minimal-80.zip"
          ],
          "task_count": 52,
          "steps_per_task": 115.34615384615384,
          "tool_calls_per_task": 488.0576923076923,
          "model_responses_per_task": 66.01923076923077,
          "variant": "",
          "series": "Meta Muse Spark 1.3",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "meta",
            "benchmark_content_hash": "b8075161613a4fdc7823a469d5e66237dec8335e34fd24f4e29f5c0b79c68d6e",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "0a1e9a0b6132d82a4ebefed299f1bc30ed8de7436d8df4ee793ce90d8719791a",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 102,
          "entry_name": "Meta Muse Spark 1.3 · low · svc_81_e35f89",
          "model": "Meta Muse Spark 1.3",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.33679122078923457,
          "time_per_task_sec": 2280.765444932827,
          "average_time_per_task_sec": 2280.765444932827,
          "median_time_per_task_sec": 1926.1888315840001,
          "cost_usd": 0.5028513153846154,
          "turns_per_task": 80.73076923076923,
          "release_date": "2026-09-02",
          "average_output_tokens": 311.29966650786093,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_81_e35f89",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-osworld2-k52-low-81.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "28af1b3bda68304fc1f1174413c712c2eb131160c36abe0e46a46f3e63074de9",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "reference_estimate",
          "cost_label": "API estimate · no cache",
          "cost_note": "API-equivalent estimate using recorded tokens and no input caching; not historical billing. Current OpenRouter public API-equivalent token price interval, not historical billing; lower assumes all input cached, upper none; excludes other charges",
          "cost_range_min_usd": 0.01498279276923077,
          "cost_range_max_usd": 0.5028513153846154,
          "cost_coverage_tasks": 52,
          "cost_total_tasks": 52,
          "cost_source": "https://openrouter.ai/api/v1/models (snapshot 2026-09-13T17:46:52Z)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.3-contributor",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-02",
            "release_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "access_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "notes": "Proprietary model available through Muse Code and Meta Model API.",
            "model": "Meta Muse Spark 1.3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_81_e35f89"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-osworld2-k52-low-81.zip"
          ],
          "task_count": 52,
          "steps_per_task": 130.5,
          "tool_calls_per_task": 655.0769230769231,
          "model_responses_per_task": 80.73076923076923,
          "variant": "",
          "series": "Meta Muse Spark 1.3",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "meta",
            "benchmark_content_hash": "b8075161613a4fdc7823a469d5e66237dec8335e34fd24f4e29f5c0b79c68d6e",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "0a1e9a0b6132d82a4ebefed299f1bc30ed8de7436d8df4ee793ce90d8719791a",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 103,
          "entry_name": "Meta Muse Spark 1.3 · medium · svc_82_a48586",
          "model": "Meta Muse Spark 1.3",
          "effort": "medium",
          "effort_rank": 2,
          "model_type": "closed",
          "performance": 0.3230579874951483,
          "time_per_task_sec": 2292.096530432519,
          "average_time_per_task_sec": 2292.096530432519,
          "median_time_per_task_sec": 2127.9897116255,
          "cost_usd": 0.5883309173076923,
          "turns_per_task": 89.75,
          "release_date": "2026-09-02",
          "average_output_tokens": 458.479537175916,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_82_a48586",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-osworld2-k52-medium-82.zip",
          "cost_basis": "unavailable",
          "notes": "Resumed after earlier interruptions; final archived event is run_done at 2026-09-09T19:09:11.302490+00:00, with all 52 planned tasks scored and timed. Earlier run_failed events are retained as history, not the final outcome.",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "0e02ab5636a31f119f1fbaab2ee8322697e5e57b2966cc4e59f79eb3efb45665",
          "run_terminal_events": "[\"run_done\",\"run_done\",\"run_failed\",\"run_done\",\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "reference_estimate",
          "cost_label": "API estimate · no cache",
          "cost_note": "API-equivalent estimate using recorded tokens and no input caching; not historical billing. Current OpenRouter public API-equivalent token price interval, not historical billing; lower assumes all input cached, upper none; excludes other charges",
          "cost_range_min_usd": 0.019831731884615383,
          "cost_range_max_usd": 0.5883309173076923,
          "cost_coverage_tasks": 52,
          "cost_total_tasks": 52,
          "cost_source": "https://openrouter.ai/api/v1/models (snapshot 2026-09-13T17:46:52Z)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.3-contributor",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-02",
            "release_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "access_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "notes": "Proprietary model available through Muse Code and Meta Model API.",
            "model": "Meta Muse Spark 1.3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_82_a48586"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-osworld2-k52-medium-82.zip"
          ],
          "task_count": 52,
          "steps_per_task": 149.80769230769232,
          "tool_calls_per_task": 832.4423076923077,
          "model_responses_per_task": 89.75,
          "variant": "",
          "series": "Meta Muse Spark 1.3",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "meta",
            "benchmark_content_hash": "b8075161613a4fdc7823a469d5e66237dec8335e34fd24f4e29f5c0b79c68d6e",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "0a1e9a0b6132d82a4ebefed299f1bc30ed8de7436d8df4ee793ce90d8719791a",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 105,
          "entry_name": "Meta Muse Spark 1.3 · xhigh · svc_84_e2ef21",
          "model": "Meta Muse Spark 1.3",
          "effort": "xhigh",
          "effort_rank": 4,
          "model_type": "closed",
          "performance": 0.3215834123722564,
          "time_per_task_sec": 2260.9855525490384,
          "average_time_per_task_sec": 2260.9855525490384,
          "median_time_per_task_sec": 2221.476171339,
          "cost_usd": 0.639914848076923,
          "turns_per_task": 93.96153846153847,
          "release_date": "2026-09-02",
          "average_output_tokens": 560.0970118706508,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_84_e2ef21",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-osworld2-k52-xhigh-84.zip",
          "cost_basis": "unavailable",
          "notes": "Resumed after earlier interruptions; final archived event is run_done at 2026-09-10T01:35:24.715393+00:00, with all 52 planned tasks scored and timed. Earlier run_failed events are retained as history, not the final outcome.",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "63522cc3f5e712dfcbc2bfd03a01f9595f0b2c75f0a35215b778176c73b38e92",
          "run_terminal_events": "[\"run_failed\",\"run_done\",\"run_done\",\"run_done\",\"run_failed\",\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "reference_estimate",
          "cost_label": "API estimate · no cache",
          "cost_note": "API-equivalent estimate using recorded tokens and no input caching; not historical billing. Current OpenRouter public API-equivalent token price interval, not historical billing; lower assumes all input cached, upper none; excludes other charges",
          "cost_range_min_usd": 0.02311330203846154,
          "cost_range_max_usd": 0.639914848076923,
          "cost_coverage_tasks": 52,
          "cost_total_tasks": 52,
          "cost_source": "https://openrouter.ai/api/v1/models (snapshot 2026-09-13T17:46:52Z)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.3-contributor",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-02",
            "release_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "access_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "notes": "Proprietary model available through Muse Code and Meta Model API.",
            "model": "Meta Muse Spark 1.3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_84_e2ef21"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-osworld2-k52-xhigh-84.zip"
          ],
          "task_count": 52,
          "steps_per_task": 147.28846153846155,
          "tool_calls_per_task": 832.75,
          "model_responses_per_task": 93.96153846153847,
          "variant": "",
          "series": "Meta Muse Spark 1.3",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "meta",
            "benchmark_content_hash": "b8075161613a4fdc7823a469d5e66237dec8335e34fd24f4e29f5c0b79c68d6e",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "8283282901039159ea67730ec08370a0291d616e884a224a7a70e6d518c1ec9d",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 101,
          "entry_name": "Meta Muse Spark 1.3 · high · svc_83_03ee99",
          "model": "Meta Muse Spark 1.3",
          "effort": "high",
          "effort_rank": 3,
          "model_type": "closed",
          "performance": 0.31459525102905256,
          "time_per_task_sec": 2179.275142096404,
          "average_time_per_task_sec": 2179.275142096404,
          "median_time_per_task_sec": 2221.1308896385003,
          "cost_usd": 0.5985068807692308,
          "turns_per_task": 89.82692307692308,
          "release_date": "2026-09-02",
          "average_output_tokens": 492.5165917362449,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_83_03ee99",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-osworld2-k52-high-83.zip",
          "cost_basis": "unavailable",
          "notes": "Resumed after earlier interruptions; final archived event is run_done at 2026-09-09T21:41:29.284094+00:00, with all 52 planned tasks scored and timed. Earlier run_failed events are retained as history, not the final outcome.",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "c2f07298ec9cfcd8d5d80eefd8cfb6327b2c0e6258067c42f6476de7277cb4ce",
          "run_terminal_events": "[\"run_done\",\"run_failed\",\"run_done\",\"run_done\",\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "reference_estimate",
          "cost_label": "API estimate · no cache",
          "cost_note": "API-equivalent estimate using recorded tokens and no input caching; not historical billing. Current OpenRouter public API-equivalent token price interval, not historical billing; lower assumes all input cached, upper none; excludes other charges",
          "cost_range_min_usd": 0.020641422615384614,
          "cost_range_max_usd": 0.5985068807692308,
          "cost_coverage_tasks": 52,
          "cost_total_tasks": 52,
          "cost_source": "https://openrouter.ai/api/v1/models (snapshot 2026-09-13T17:46:52Z)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "meta/muse-spark-1.3-contributor",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-02",
            "release_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "access_source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
            "notes": "Proprietary model available through Muse Code and Meta Model API.",
            "model": "Meta Muse Spark 1.3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_83_03ee99"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-musespark13-osworld2-k52-high-83.zip"
          ],
          "task_count": 52,
          "steps_per_task": 151.21153846153845,
          "tool_calls_per_task": 754.9807692307693,
          "model_responses_per_task": 89.82692307692308,
          "variant": "",
          "series": "Meta Muse Spark 1.3",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "meta",
            "benchmark_content_hash": "b8075161613a4fdc7823a469d5e66237dec8335e34fd24f4e29f5c0b79c68d6e",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "0a1e9a0b6132d82a4ebefed299f1bc30ed8de7436d8df4ee793ce90d8719791a",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "contributor",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 141,
          "entry_name": "Kimi K3 · max · svc_147_d5291e",
          "model": "Kimi K3",
          "effort": "max",
          "effort_rank": 5,
          "model_type": "open",
          "performance": 0.28506236962992637,
          "time_per_task_sec": 4825.08071171677,
          "average_time_per_task_sec": 4825.08071171677,
          "median_time_per_task_sec": 3565.2872566635,
          "cost_usd": 5.624738163461539,
          "turns_per_task": 60.78846153846154,
          "release_date": "2026-07-16",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_147_d5291e",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/00d7d65cb6459aaf02767e8ab6b3ad419f2c66b1/pranjal-kimi-k3-parallel-osworld2-k52-max-500steps-modal-v1-147-resumed-20260919.zip",
          "cost_basis": "recorded_provider_usage_cost",
          "notes": "Completed September 19 resume of evaluation 147; replaces its historical billing-affected snapshot for comparison. Only the 15 authorized billing-interrupted task/seed pairs were rerun; 37 original results unchanged. All 52 final tasks complete, with recorded cost and no HTTP 402 errors. Original frozen plan and submission unchanged. Prior interrupted attempts retained separately and overlap historical archives; do not sum snapshots as independent evaluations. Completed max-effort resume; the earlier billing-interrupted snapshot remains in the downloadable CSV. One final task reports more reasoning tokens than total output tokens; raw provider counters are retained and the impossible derived non-reasoning count is left unknown.",
          "quality_flags": "{\"Kimi_finish_rejected_response_usage_included_but_tool_bodies_unlogged\":10,\"Kimi_noncompletion_reply_usage_included_body_unlogged\":1,\"thinking_exceeds_generated\":1}",
          "selected_task_seed_set_sha256": "f8b2f5cb1f9c7edf87e0f13c01cdf7281aae31ecc8834407cd409fd144e7e61a",
          "run_terminal_events": "[\"run_done\",\"run_failed\",\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 5.624738163461539,
          "cost_kind": "recorded",
          "cost_label": "Recorded usage",
          "cost_note": "recorded_provider_usage_cost. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 52,
          "cost_total_tasks": 52,
          "cost_source": "recorded_provider_usage_cost",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/00d7d65cb6459aaf02767e8ab6b3ad419f2c66b1/pranjal-kimi-k3-parallel-osworld2-k52-max-500steps-modal-v1-147-resumed-20260919.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "open",
            "release_date": "2026-07-16",
            "release_source": "https://www.kimi.com/en/blog/kimi-k3",
            "access_source": "https://huggingface.co/moonshotai/Kimi-K3",
            "notes": "Public model launch July 16; downloadable weights followed later. API use in the evaluation does not make this a closed-weight model.",
            "model": "Kimi K3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_147_d5291e"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/00d7d65cb6459aaf02767e8ab6b3ad419f2c66b1/pranjal-kimi-k3-parallel-osworld2-k52-max-500steps-modal-v1-147-resumed-20260919.zip"
          ],
          "task_count": 52,
          "steps_per_task": 66.38461538461539,
          "tool_calls_per_task": 386.5576923076923,
          "model_responses_per_task": 60.78846153846154,
          "variant": "",
          "series": "Kimi K3",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "kimi_parallel",
            "benchmark_content_hash": "681050889b7b296c9e48f6ec013c3bc897f38cccd9407fdcb0fe2dace82f6e26",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "34761619b59911dddf1748656f82a27a24896a268da28e7192a6407cad0f67b7",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 139,
          "entry_name": "Kimi K3 · high · svc_145_4091f6",
          "model": "Kimi K3",
          "effort": "high",
          "effort_rank": 3,
          "model_type": "open",
          "performance": 0.2259537286565601,
          "time_per_task_sec": 3733.9232542997497,
          "average_time_per_task_sec": 3733.9232542997497,
          "median_time_per_task_sec": 2263.6440786880003,
          "cost_usd": 4.517302015384615,
          "turns_per_task": 59.03846153846154,
          "release_date": "2026-07-16",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_145_4091f6",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/00d7d65cb6459aaf02767e8ab6b3ad419f2c66b1/pranjal-kimi-k3-parallel-osworld2-k52-high-500steps-modal-v1-145-resumed-20260919.zip",
          "cost_basis": "recorded_provider_usage_cost",
          "notes": "Completed September 19 resume of evaluation 145; replaces its historical billing-affected snapshot for comparison. Only the 2 authorized billing-interrupted task/seed pairs were rerun; 50 original results unchanged. All 52 final tasks complete, with recorded cost and no HTTP 402 errors. Original frozen plan and submission unchanged. Prior interrupted attempts retained separately and overlap historical archives; do not sum snapshots as independent evaluations. Completed high-effort resume; the earlier billing-interrupted snapshot remains in the downloadable CSV.",
          "quality_flags": "{\"Kimi_finish_rejected_response_usage_included_but_tool_bodies_unlogged\":5}",
          "selected_task_seed_set_sha256": "fe1fcfebd5997098dcec14795178ab1007637bed442640e0b958e32e0292c6f3",
          "run_terminal_events": "[\"run_failed\",\"run_done\",\"run_failed\",\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 4.517302015384615,
          "cost_kind": "recorded",
          "cost_label": "Recorded usage",
          "cost_note": "recorded_provider_usage_cost. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 52,
          "cost_total_tasks": 52,
          "cost_source": "recorded_provider_usage_cost",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/00d7d65cb6459aaf02767e8ab6b3ad419f2c66b1/pranjal-kimi-k3-parallel-osworld2-k52-high-500steps-modal-v1-145-resumed-20260919.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "open",
            "release_date": "2026-07-16",
            "release_source": "https://www.kimi.com/en/blog/kimi-k3",
            "access_source": "https://huggingface.co/moonshotai/Kimi-K3",
            "notes": "Public model launch July 16; downloadable weights followed later. API use in the evaluation does not make this a closed-weight model.",
            "model": "Kimi K3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_145_4091f6"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/00d7d65cb6459aaf02767e8ab6b3ad419f2c66b1/pranjal-kimi-k3-parallel-osworld2-k52-high-500steps-modal-v1-145-resumed-20260919.zip"
          ],
          "task_count": 52,
          "steps_per_task": 58.76923076923077,
          "tool_calls_per_task": 350.0576923076923,
          "model_responses_per_task": 59.03846153846154,
          "variant": "",
          "series": "Kimi K3",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "kimi_parallel",
            "benchmark_content_hash": "681050889b7b296c9e48f6ec013c3bc897f38cccd9407fdcb0fe2dace82f6e26",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "34761619b59911dddf1748656f82a27a24896a268da28e7192a6407cad0f67b7",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 99,
          "entry_name": "Kimi K3 · low · svc_146_518434",
          "model": "Kimi K3",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "open",
          "performance": 0.19190740246556307,
          "time_per_task_sec": 1860.850342020019,
          "average_time_per_task_sec": 1860.850342020019,
          "median_time_per_task_sec": 1446.5695975854999,
          "cost_usd": 2.088021098076923,
          "turns_per_task": 49.17307692307692,
          "release_date": "2026-07-16",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_146_518434",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9d7e011c785d433d2329ec51785aa3b6ebbee312/pranjal-kimi-k3-parallel-osworld2-k52-low-500steps-modal-v1-146.zip",
          "cost_basis": "recorded_provider_usage_cost",
          "notes": "Resumed after earlier interruptions; final archived event is run_done at 2026-09-13T12:11:51.227606+00:00, with all 52 planned tasks scored and timed. Earlier run_failed events are retained as history, not the final outcome.",
          "quality_flags": "{\"Kimi_finish_rejected_response_usage_included_but_tool_bodies_unlogged\":1}",
          "selected_task_seed_set_sha256": "9defc3b57a1293b37db7236b7bf8f7d6e7473ec74182fff6ae3757a5993c8c7a",
          "run_terminal_events": "[\"run_failed\",\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 2.088021098076923,
          "cost_kind": "recorded",
          "cost_label": "Recorded usage",
          "cost_note": "recorded_provider_usage_cost. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 52,
          "cost_total_tasks": 52,
          "cost_source": "recorded_provider_usage_cost",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9d7e011c785d433d2329ec51785aa3b6ebbee312/pranjal-kimi-k3-parallel-osworld2-k52-low-500steps-modal-v1-146.zip",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "open",
            "release_date": "2026-07-16",
            "release_source": "https://www.kimi.com/en/blog/kimi-k3",
            "access_source": "https://huggingface.co/moonshotai/Kimi-K3",
            "notes": "Public model launch July 16; downloadable weights followed later. API use in the evaluation does not make this a closed-weight model.",
            "model": "Kimi K3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_146_518434"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9d7e011c785d433d2329ec51785aa3b6ebbee312/pranjal-kimi-k3-parallel-osworld2-k52-low-500steps-modal-v1-146.zip"
          ],
          "task_count": 52,
          "steps_per_task": 47.25,
          "tool_calls_per_task": 313.75,
          "model_responses_per_task": 49.17307692307692,
          "variant": "",
          "series": "Kimi K3",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "kimi_parallel",
            "benchmark_content_hash": "681050889b7b296c9e48f6ec013c3bc897f38cccd9407fdcb0fe2dace82f6e26",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "34761619b59911dddf1748656f82a27a24896a268da28e7192a6407cad0f67b7",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 107,
          "entry_name": "MiniMax M3 · thinking_on · svc_78_35a939",
          "model": "MiniMax M3",
          "effort": "thinking_on",
          "effort_rank": 6,
          "model_type": "open",
          "performance": 0.04102336976043484,
          "time_per_task_sec": 1673.2428303149807,
          "average_time_per_task_sec": 1673.2428303149807,
          "median_time_per_task_sec": 1483.140117951,
          "cost_usd": null,
          "turns_per_task": 98.61538461538461,
          "release_date": "2026-06-01",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_78_35a939",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-minimax-m3-osworld2-k52-thinking-on-78.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{\"output_truncated_at_400_characters_and_usage_not_logged\":52}",
          "selected_task_seed_set_sha256": "771c0d488bd997d106d05aff124b45688662a8727ece87b0617c01fc67eee97a",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "unavailable",
          "cost_label": "Usage not logged",
          "cost_note": "Standard API rates through 512K context; above 512K, input/output/cache rates double. Priority pricing differs. Archived responses do not retain token usage, so unit prices cannot yield a per-task total.",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 0,
          "cost_total_tasks": 52,
          "cost_source": "https://platform.minimax.io/subscribe/token-plan?tab=api-enterprise (verified 2026-09-16)",
          "cost_source_url": "https://platform.minimax.io/subscribe/token-plan?tab=api-enterprise",
          "cost_pricing_model": "minimax/minimax-m3",
          "cost_unit_prices": {
            "model_id": "minimax/minimax-m3",
            "input": 0.3,
            "cached_input": 0.06,
            "output": 1.2,
            "estimate_from_usage": false,
            "unit_note": "standard · ≤512K context",
            "source_url": "https://platform.minimax.io/subscribe/token-plan?tab=api-enterprise",
            "source_page": "https://www.minimax.io/blog/minimax-m3",
            "note": "Standard API rates through 512K context; above 512K, input/output/cache rates double. Priority pricing differs. Archived responses do not retain token usage, so unit prices cannot yield a per-task total."
          },
          "model_metadata": {
            "model_type": "open",
            "release_date": "2026-06-01",
            "release_source": "https://www.minimax.io/blog/minimax-m3",
            "access_source": "https://huggingface.co/MiniMaxAI/MiniMax-M3",
            "notes": "Public model launch June 1; downloadable weights followed later under the MiniMax Community License. The hyphenated legacy label identifies the same model.",
            "model": "MiniMax M3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_78_35a939"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-minimax-m3-osworld2-k52-thinking-on-78.zip"
          ],
          "task_count": 52,
          "steps_per_task": 97.5576923076923,
          "tool_calls_per_task": 407.3269230769231,
          "model_responses_per_task": 98.61538461538461,
          "variant": "",
          "series": "MiniMax M3",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "minimax_m3",
            "benchmark_content_hash": "b8075161613a4fdc7823a469d5e66237dec8335e34fd24f4e29f5c0b79c68d6e",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "0a1e9a0b6132d82a4ebefed299f1bc30ed8de7436d8df4ee793ce90d8719791a",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 106,
          "entry_name": "MiniMax M3 · thinking_off · svc_77_abf121",
          "model": "MiniMax M3",
          "effort": "thinking_off",
          "effort_rank": 0,
          "model_type": "open",
          "performance": 0.032367307692307695,
          "time_per_task_sec": 1446.9359536971538,
          "average_time_per_task_sec": 1446.9359536971538,
          "median_time_per_task_sec": 1332.4022616164998,
          "cost_usd": null,
          "turns_per_task": 98.65384615384616,
          "release_date": "2026-06-01",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_77_abf121",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-minimax-m3-osworld2-k52-thinking-off-77.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{\"output_truncated_at_400_characters_and_usage_not_logged\":52}",
          "selected_task_seed_set_sha256": "76b48b209e9a2b9758f6f814380f66f4d3c99e441e5e7e919b01c247a2eb578b",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "unavailable",
          "cost_label": "Usage not logged",
          "cost_note": "Standard API rates through 512K context; above 512K, input/output/cache rates double. Priority pricing differs. Archived responses do not retain token usage, so unit prices cannot yield a per-task total.",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 0,
          "cost_total_tasks": 52,
          "cost_source": "https://platform.minimax.io/subscribe/token-plan?tab=api-enterprise (verified 2026-09-16)",
          "cost_source_url": "https://platform.minimax.io/subscribe/token-plan?tab=api-enterprise",
          "cost_pricing_model": "minimax/minimax-m3",
          "cost_unit_prices": {
            "model_id": "minimax/minimax-m3",
            "input": 0.3,
            "cached_input": 0.06,
            "output": 1.2,
            "estimate_from_usage": false,
            "unit_note": "standard · ≤512K context",
            "source_url": "https://platform.minimax.io/subscribe/token-plan?tab=api-enterprise",
            "source_page": "https://www.minimax.io/blog/minimax-m3",
            "note": "Standard API rates through 512K context; above 512K, input/output/cache rates double. Priority pricing differs. Archived responses do not retain token usage, so unit prices cannot yield a per-task total."
          },
          "model_metadata": {
            "model_type": "open",
            "release_date": "2026-06-01",
            "release_source": "https://www.minimax.io/blog/minimax-m3",
            "access_source": "https://huggingface.co/MiniMaxAI/MiniMax-M3",
            "notes": "Public model launch June 1; downloadable weights followed later under the MiniMax Community License. The hyphenated legacy label identifies the same model.",
            "model": "MiniMax M3",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_77_abf121"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-minimax-m3-osworld2-k52-thinking-off-77.zip"
          ],
          "task_count": 52,
          "steps_per_task": 98.53846153846153,
          "tool_calls_per_task": 368.61538461538464,
          "model_responses_per_task": 98.65384615384616,
          "variant": "",
          "series": "MiniMax M3",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "minimax_m3",
            "benchmark_content_hash": "b8075161613a4fdc7823a469d5e66237dec8335e34fd24f4e29f5c0b79c68d6e",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "0a1e9a0b6132d82a4ebefed299f1bc30ed8de7436d8df4ee793ce90d8719791a",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 125,
          "entry_name": "GLM-5V Turbo · thinking_off · svc_88_3a6bc6",
          "model": "GLM-5V Turbo",
          "effort": "thinking_off",
          "effort_rank": 0,
          "model_type": "closed",
          "performance": 0.015319230769230769,
          "time_per_task_sec": 1979.3570428305768,
          "average_time_per_task_sec": 1979.3570428305768,
          "median_time_per_task_sec": 1959.068403869,
          "cost_usd": 0.7394585461538461,
          "turns_per_task": 48.38461538461539,
          "release_date": "2026-04-01",
          "average_output_tokens": 590.226947535771,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_88_3a6bc6",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61afda292c109e49aaa87f9dfc35f4e8bc5363a9/ljang-glm5v-turbo-osworld2-k52-thinking-off-pillow-fixed-88.zip",
          "cost_basis": "unavailable",
          "notes": "Publisher-designated corrected thinking-off/Pillow-fixed rerun; Pillow 11.3.0 installation is present in init.log. Recorded dollar cost and reasoning/cache token breakdown are unavailable, not zero. Separate reference-cost bounds use the cited OpenRouter rate snapshot and assume all versus no input cached; they are not billing.",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "3c39e78198063ed75c5220133969c410e04dfd4b161460724f9fdde11cd38a5d",
          "run_terminal_events": "[\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "reference_estimate",
          "cost_label": "API estimate · no cache",
          "cost_note": "API-equivalent estimate using recorded tokens and no input caching; not historical billing. Current OpenRouter public API-equivalent token price interval, not historical billing; lower assumes all input cached, upper none; excludes other charges",
          "cost_range_min_usd": 0.23927700153846151,
          "cost_range_max_usd": 0.7394585461538461,
          "cost_coverage_tasks": 52,
          "cost_total_tasks": 52,
          "cost_source": "https://openrouter.ai/api/v1/models (snapshot 2026-09-15T20:56:37.449456+00:00)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "z-ai/glm-5v-turbo",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-04-01",
            "release_source": "https://docs.z.ai/release-notes/new-released",
            "access_source": "https://docs.z.ai/guides/vlm/glm-5v-turbo",
            "notes": "The publisher's release notes date this API model April 1; the GLM-V repository posted its announcement April 2. GLM-5V-Turbo is not one of the downloadable GLM-V models; the GLM family name does not establish open weights.",
            "model": "GLM-5V Turbo",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_88_3a6bc6"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/61afda292c109e49aaa87f9dfc35f4e8bc5363a9/ljang-glm5v-turbo-osworld2-k52-thinking-off-pillow-fixed-88.zip"
          ],
          "task_count": 52,
          "steps_per_task": 49.15384615384615,
          "tool_calls_per_task": 79.34615384615384,
          "model_responses_per_task": 48.38461538461539,
          "variant": "",
          "series": "GLM-5V Turbo",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "glm5v_turbo",
            "benchmark_content_hash": "b8075161613a4fdc7823a469d5e66237dec8335e34fd24f4e29f5c0b79c68d6e",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "34761619b59911dddf1748656f82a27a24896a268da28e7192a6407cad0f67b7",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 90,
          "entry_name": "GLM-5V Turbo · thinking_on · svc_79_c47638",
          "model": "GLM-5V Turbo",
          "effort": "thinking_on",
          "effort_rank": 6,
          "model_type": "closed",
          "performance": 0.014549999999999999,
          "time_per_task_sec": 2641.9071493502115,
          "average_time_per_task_sec": 2641.9071493502115,
          "median_time_per_task_sec": 2434.7209227935,
          "cost_usd": 0.7250202,
          "turns_per_task": 47.11538461538461,
          "release_date": "2026-04-01",
          "average_output_tokens": 602.4673469387756,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_79_c47638",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-glm5v-turbo-osworld2-k52-thinking-on-79.zip",
          "cost_basis": "unavailable",
          "notes": "",
          "quality_flags": "{}",
          "selected_task_seed_set_sha256": "b9f456d5f061ba5b03827f5c390d6351e27e858e706f64de1cb8a143fe74e3e3",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": null,
          "cost_kind": "reference_estimate",
          "cost_label": "API estimate · no cache",
          "cost_note": "API-equivalent estimate using recorded tokens and no input caching; not historical billing. Current OpenRouter public API-equivalent token price interval, not historical billing; lower assumes all input cached, upper none; excludes other charges",
          "cost_range_min_usd": 0.23583757846153844,
          "cost_range_max_usd": 0.7250202,
          "cost_coverage_tasks": 52,
          "cost_total_tasks": 52,
          "cost_source": "https://openrouter.ai/api/v1/models (snapshot 2026-09-13T17:46:52Z)",
          "cost_source_url": "https://openrouter.ai/api/v1/models",
          "cost_pricing_model": "z-ai/glm-5v-turbo",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-04-01",
            "release_source": "https://docs.z.ai/release-notes/new-released",
            "access_source": "https://docs.z.ai/guides/vlm/glm-5v-turbo",
            "notes": "The publisher's release notes date this API model April 1; the GLM-V repository posted its announcement April 2. GLM-5V-Turbo is not one of the downloadable GLM-V models; the GLM family name does not establish open weights.",
            "model": "GLM-5V Turbo",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_79_c47638"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/60887ee9e16c7fead81dabad9bfb60b1471d89c5/ljang-glm5v-turbo-osworld2-k52-thinking-on-79.zip"
          ],
          "task_count": 52,
          "steps_per_task": 47.75,
          "tool_calls_per_task": 91.75,
          "model_responses_per_task": 47.11538461538461,
          "variant": "",
          "series": "GLM-5V Turbo",
          "source_metadata": {
            "split_dset_name": "osworld2-k52",
            "benchmark_name": "osworld2-k52",
            "benchmark_version": "2026.06.24",
            "template_name": "glm5v_turbo",
            "benchmark_content_hash": "b8075161613a4fdc7823a469d5e66237dec8335e34fd24f4e29f5c0b79c68d6e",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "0a1e9a0b6132d82a4ebefed299f1bc30ed8de7436d8df4ee793ce90d8719791a",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[39600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        }
      ],
      "excluded_evaluations": [],
      "supporting_rows": 14,
      "task_counts": [
        52
      ],
      "paper_subset": {
        "label": "OSWorld 2.0",
        "selected_tasks": 52,
        "eligible_tasks": 63
      }
    },
    "cua-world-long-k26": {
      "name": "cua-world-long-k26",
      "label": "CUA-World · 26 tasks",
      "href": "dataset-cua-world-long-k26.html",
      "records": [
        {
          "entry_id": 3,
          "entry_name": "GPT-6 Astra · xhigh · svc_139_c67de8",
          "model": "GPT-6 Astra",
          "effort": "xhigh",
          "effort_rank": 4,
          "model_type": "closed",
          "performance": 0.9298076923076922,
          "time_per_task_sec": 1515.4057694185383,
          "average_time_per_task_sec": 1515.4057694185383,
          "median_time_per_task_sec": 825.901089452,
          "cost_usd": 15.061029846153845,
          "turns_per_task": null,
          "release_date": "2026-09-03",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_139_c67de8",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9b541e31ddb6c7ffcfde1f460f5c2d449f625072/pranjal-gpt6astra-cua-world-k26-xhigh-500steps-modal-v1-139.zip",
          "cost_basis": "Astra_standard_short_context_API_equivalent_not_subscription_bill",
          "notes": "Cost is short-context API-equivalent usage, not a subscription payment; full tool/response trace unavailable in CLI export.",
          "quality_flags": "{\"Astra_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":26,\"codex_model_response_count_and_full_tool_calls_unavailable\":26}",
          "selected_task_seed_set_sha256": "3ad10d87e90f46f5dfad385b01ff73491eacd7b44d59d60507091208face129d",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 15.061029846153845,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Astra_standard_short_context_API_equivalent_not_subscription_bill. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 26,
          "cost_total_tasks": 26,
          "cost_source": "HF COST-pranjal-eval137.json pinned pricing snapshot; input/cache/write/output=10/1/12.5/50 USD per million",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9b541e31ddb6c7ffcfde1f460f5c2d449f625072/COST-pranjal-eval137.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-03",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-6-astra",
            "notes": "OpenAI's September 3, 2026 changelog announces GPT-6 Astra. Proprietary hosted model.",
            "model": "GPT-6 Astra",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_139_c67de8"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9b541e31ddb6c7ffcfde1f460f5c2d449f625072/pranjal-gpt6astra-cua-world-k26-xhigh-500steps-modal-v1-139.zip"
          ],
          "task_count": 26,
          "steps_per_task": 103.11538461538461,
          "tool_calls_per_task": 528.1538461538462,
          "model_responses_per_task": null,
          "variant": "",
          "series": "GPT-6 Astra",
          "source_metadata": {
            "split_dset_name": "cua-world-long-k26",
            "benchmark_name": "cua-world-long-k26",
            "benchmark_version": "0.6",
            "template_name": "codex_cli",
            "benchmark_content_hash": "b1f68d7d7a0790f55a8678922250a6f0334d6917b7128ac6ad007bf1e3a296de",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "34761619b59911dddf1748656f82a27a24896a268da28e7192a6407cad0f67b7",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[21600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "16"
          }
        },
        {
          "entry_id": 2,
          "entry_name": "GPT-6 Astra · low · svc_137_284111",
          "model": "GPT-6 Astra",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.9028846153846153,
          "time_per_task_sec": 806.8412924521155,
          "average_time_per_task_sec": 806.8412924521155,
          "median_time_per_task_sec": 639.678005643,
          "cost_usd": 9.286949615384616,
          "turns_per_task": null,
          "release_date": "2026-09-03",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_137_284111",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9b541e31ddb6c7ffcfde1f460f5c2d449f625072/pranjal-gpt6astra-cua-world-k26-low-500steps-modal-v1-137.zip",
          "cost_basis": "Astra_standard_short_context_API_equivalent_not_subscription_bill",
          "notes": "Cost is short-context API-equivalent usage, not a subscription payment; full tool/response trace unavailable in CLI export.",
          "quality_flags": "{\"Astra_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":26,\"codex_model_response_count_and_full_tool_calls_unavailable\":26}",
          "selected_task_seed_set_sha256": "d87d758b464450c51024db1d28a16449824bebab3d8897d57ee026cca5bbf982",
          "run_terminal_events": "[]",
          "row_type": "evaluation",
          "source_cost_usd": 9.286949615384616,
          "cost_kind": "source_estimate",
          "cost_label": "Logged estimate",
          "cost_note": "Astra_standard_short_context_API_equivalent_not_subscription_bill. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 26,
          "cost_total_tasks": 26,
          "cost_source": "HF COST-pranjal-eval137.json pinned pricing snapshot; input/cache/write/output=10/1/12.5/50 USD per million",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9b541e31ddb6c7ffcfde1f460f5c2d449f625072/COST-pranjal-eval137.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-03",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-6-astra",
            "notes": "OpenAI's September 3, 2026 changelog announces GPT-6 Astra. Proprietary hosted model.",
            "model": "GPT-6 Astra",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_137_284111"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/9b541e31ddb6c7ffcfde1f460f5c2d449f625072/pranjal-gpt6astra-cua-world-k26-low-500steps-modal-v1-137.zip"
          ],
          "task_count": 26,
          "steps_per_task": 68.34615384615384,
          "tool_calls_per_task": 220.69230769230768,
          "model_responses_per_task": null,
          "variant": "",
          "series": "GPT-6 Astra",
          "source_metadata": {
            "split_dset_name": "cua-world-long-k26",
            "benchmark_name": "cua-world-long-k26",
            "benchmark_version": "0.6",
            "template_name": "codex_cli",
            "benchmark_content_hash": "b1f68d7d7a0790f55a8678922250a6f0334d6917b7128ac6ad007bf1e3a296de",
            "track_name": "api-cpu-agents1",
            "measurement_hash": "34761619b59911dddf1748656f82a27a24896a268da28e7192a6407cad0f67b7",
            "eval_algorithm": "shared-agent-vllm@2",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[21600.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-remote",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "13"
          }
        }
      ],
      "excluded_evaluations": [],
      "supporting_rows": 0,
      "task_counts": [
        26
      ],
      "paper_subset": {
        "label": "CUA-World",
        "selected_tasks": 26,
        "eligible_tasks": 143
      }
    },
    "mypcbench-energy38": {
      "name": "mypcbench-energy38",
      "label": "MyPCBench · 38 tasks",
      "href": "dataset-mypcbench-energy38.html",
      "records": [
        {
          "entry_id": 146,
          "entry_name": "GPT-6 Astra · xhigh · svc_96_fee517",
          "model": "GPT-6 Astra",
          "effort": "xhigh",
          "effort_rank": 4,
          "model_type": "closed",
          "performance": 0.9355263157894737,
          "time_per_task_sec": 619.1620519224474,
          "average_time_per_task_sec": 619.1620519224474,
          "median_time_per_task_sec": 654.4426635405,
          "cost_usd": 4.809898789473684,
          "turns_per_task": null,
          "release_date": "2026-09-03",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_96_fee517",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/447ff4cd77230ee06ee5b0e864873c43ceb0138c/pranjal-gpt6astra-codex-mypcbench-energy38-xhigh-100steps-no-preload-modal-v1-96.zip",
          "cost_basis": "published_API_rate_estimate_from_logged_usage_and_cache_not_recorded_charge",
          "notes": "Published HF PR25, merged before inventory. GPT-6 Astra via Codex; 100 CUA batches, 7200-second timeout, no-preload track, 38 original partial-credit scores retained. Cost is an API-equivalent estimate from complete logged usage and cache counts at the published September 12 reference rates, not Codex-plan spending or a current-price quote. 0/38 judge bundles lacked supplied final-response text.",
          "quality_flags": "{\"Astra_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":38,\"codex_model_response_count_and_full_tool_calls_unavailable\":38}",
          "selected_task_seed_set_sha256": "a12dc9366abca098364c91f138f081b08f0d9c97b76809333b0db9416e1d18f7",
          "run_terminal_events": "[\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 4.809898789473684,
          "cost_kind": "source_estimate",
          "cost_label": "API estimate · logged usage",
          "cost_note": "published_API_rate_estimate_from_logged_usage_and_cache_not_recorded_charge. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 38,
          "cost_total_tasks": 38,
          "cost_source": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/447ff4cd77230ee06ee5b0e864873c43ceb0138c/COST-pranjal-eval137.json (reference rates recorded 2026-09-12T05:10:02.711841+00:00; input/cache/write/output=10/1/12.5/50 USD per million; not current-price verification)",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/447ff4cd77230ee06ee5b0e864873c43ceb0138c/COST-pranjal-eval137.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-03",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-6-astra",
            "notes": "OpenAI's September 3, 2026 changelog announces GPT-6 Astra. Proprietary hosted model.",
            "model": "GPT-6 Astra",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_96_fee517"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/447ff4cd77230ee06ee5b0e864873c43ceb0138c/pranjal-gpt6astra-codex-mypcbench-energy38-xhigh-100steps-no-preload-modal-v1-96.zip"
          ],
          "task_count": 38,
          "steps_per_task": 42.13157894736842,
          "tool_calls_per_task": 123.57894736842105,
          "model_responses_per_task": null,
          "variant": "Codex",
          "series": "GPT-6 Astra · Codex",
          "source_metadata": {
            "split_dset_name": "mypcbench-energy38",
            "benchmark_name": "mypcbench-energy38",
            "benchmark_version": "0.2",
            "template_name": "codex_cli",
            "benchmark_content_hash": "77ba0d3c8521ba6f65738bf8a172690efd8d4fef35b3de9f2976ed8b66675cf5",
            "track_name": "api-cpu-agents1-no-preload",
            "measurement_hash": "c7d7714aeb30a156e468b742747141d2b0c0c834db8621f004f7705067ccf84c",
            "eval_algorithm": "shared-agent-no-preload@1",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[7200.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 144,
          "entry_name": "GPT-6 Astra · low · svc_93_cbb0ad",
          "model": "GPT-6 Astra",
          "effort": "low",
          "effort_rank": 1,
          "model_type": "closed",
          "performance": 0.8986842105263158,
          "time_per_task_sec": 514.2687531146843,
          "average_time_per_task_sec": 514.2687531146843,
          "median_time_per_task_sec": 396.02210642100005,
          "cost_usd": 4.574090263157895,
          "turns_per_task": null,
          "release_date": "2026-09-03",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_93_cbb0ad",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/447ff4cd77230ee06ee5b0e864873c43ceb0138c/pranjal-gpt6astra-codex-mypcbench-energy38-low-100steps-no-preload-modal-v1-93.zip",
          "cost_basis": "published_API_rate_estimate_from_logged_usage_and_cache_not_recorded_charge",
          "notes": "Published HF PR25, merged before inventory. GPT-6 Astra via Codex; 100 CUA batches, 7200-second timeout, no-preload track, 38 original partial-credit scores retained. Cost is an API-equivalent estimate from complete logged usage and cache counts at the published September 12 reference rates, not Codex-plan spending or a current-price quote. 2/38 judge bundles lacked supplied final-response text. Resumed after six preparation failures before agents started; 32 completed records retained and six incomplete instances reopened. No extra timed trajectories. 1 agent(s) exited 1 after the gateway finalized at the 100-batch limit; recorded verdicts unchanged.",
          "quality_flags": "{\"Astra_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":38,\"agent_exit_1_after_gateway_finalized_at_100_batches\":1,\"codex_model_response_count_and_full_tool_calls_unavailable\":38,\"judge_did_not_receive_final_response_text\":2}",
          "selected_task_seed_set_sha256": "34a30c0ced74acc7a13d2ce721dd4d596fb5209e873c2858940ab43b37ea89e3",
          "run_terminal_events": "[\"run_failed\",\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 4.574090263157895,
          "cost_kind": "source_estimate",
          "cost_label": "API estimate · logged usage",
          "cost_note": "published_API_rate_estimate_from_logged_usage_and_cache_not_recorded_charge. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 38,
          "cost_total_tasks": 38,
          "cost_source": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/447ff4cd77230ee06ee5b0e864873c43ceb0138c/COST-pranjal-eval137.json (reference rates recorded 2026-09-12T05:10:02.711841+00:00; input/cache/write/output=10/1/12.5/50 USD per million; not current-price verification)",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/447ff4cd77230ee06ee5b0e864873c43ceb0138c/COST-pranjal-eval137.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-03",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-6-astra",
            "notes": "OpenAI's September 3, 2026 changelog announces GPT-6 Astra. Proprietary hosted model.",
            "model": "GPT-6 Astra",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_93_cbb0ad"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/447ff4cd77230ee06ee5b0e864873c43ceb0138c/pranjal-gpt6astra-codex-mypcbench-energy38-low-100steps-no-preload-modal-v1-93.zip"
          ],
          "task_count": 38,
          "steps_per_task": 36.5,
          "tool_calls_per_task": 93.52631578947368,
          "model_responses_per_task": null,
          "variant": "Codex",
          "series": "GPT-6 Astra · Codex",
          "source_metadata": {
            "split_dset_name": "mypcbench-energy38",
            "benchmark_name": "mypcbench-energy38",
            "benchmark_version": "0.2",
            "template_name": "codex_cli",
            "benchmark_content_hash": "77ba0d3c8521ba6f65738bf8a172690efd8d4fef35b3de9f2976ed8b66675cf5",
            "track_name": "api-cpu-agents1-no-preload",
            "measurement_hash": "c7d7714aeb30a156e468b742747141d2b0c0c834db8621f004f7705067ccf84c",
            "eval_algorithm": "shared-agent-no-preload@1",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[7200.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 143,
          "entry_name": "GPT-6 Astra · high · svc_95_5ae456",
          "model": "GPT-6 Astra",
          "effort": "high",
          "effort_rank": 3,
          "model_type": "closed",
          "performance": 0.8863157894736843,
          "time_per_task_sec": 568.3168793826579,
          "average_time_per_task_sec": 568.3168793826579,
          "median_time_per_task_sec": 556.610181835,
          "cost_usd": 4.812435210526315,
          "turns_per_task": null,
          "release_date": "2026-09-03",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_95_5ae456",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/447ff4cd77230ee06ee5b0e864873c43ceb0138c/pranjal-gpt6astra-codex-mypcbench-energy38-high-100steps-no-preload-modal-v1-95.zip",
          "cost_basis": "published_API_rate_estimate_from_logged_usage_and_cache_not_recorded_charge",
          "notes": "Published HF PR25, merged before inventory. GPT-6 Astra via Codex; 100 CUA batches, 7200-second timeout, no-preload track, 38 original partial-credit scores retained. Cost is an API-equivalent estimate from complete logged usage and cache counts at the published September 12 reference rates, not Codex-plan spending or a current-price quote. 1/38 judge bundles lacked supplied final-response text. 1 agent(s) exited 1 after the gateway finalized at the 100-batch limit; recorded verdicts unchanged.",
          "quality_flags": "{\"Astra_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":38,\"agent_exit_1_after_gateway_finalized_at_100_batches\":1,\"codex_model_response_count_and_full_tool_calls_unavailable\":38,\"judge_did_not_receive_final_response_text\":1}",
          "selected_task_seed_set_sha256": "25eae7665b074e73215ae7b0fb528e57d8fa06f96c4bef046a97b9605ed93a79",
          "run_terminal_events": "[\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 4.812435210526315,
          "cost_kind": "source_estimate",
          "cost_label": "API estimate · logged usage",
          "cost_note": "published_API_rate_estimate_from_logged_usage_and_cache_not_recorded_charge. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 38,
          "cost_total_tasks": 38,
          "cost_source": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/447ff4cd77230ee06ee5b0e864873c43ceb0138c/COST-pranjal-eval137.json (reference rates recorded 2026-09-12T05:10:02.711841+00:00; input/cache/write/output=10/1/12.5/50 USD per million; not current-price verification)",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/447ff4cd77230ee06ee5b0e864873c43ceb0138c/COST-pranjal-eval137.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-03",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-6-astra",
            "notes": "OpenAI's September 3, 2026 changelog announces GPT-6 Astra. Proprietary hosted model.",
            "model": "GPT-6 Astra",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_95_5ae456"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/447ff4cd77230ee06ee5b0e864873c43ceb0138c/pranjal-gpt6astra-codex-mypcbench-energy38-high-100steps-no-preload-modal-v1-95.zip"
          ],
          "task_count": 38,
          "steps_per_task": 40.36842105263158,
          "tool_calls_per_task": 112.13157894736842,
          "model_responses_per_task": null,
          "variant": "Codex",
          "series": "GPT-6 Astra · Codex",
          "source_metadata": {
            "split_dset_name": "mypcbench-energy38",
            "benchmark_name": "mypcbench-energy38",
            "benchmark_version": "0.2",
            "template_name": "codex_cli",
            "benchmark_content_hash": "77ba0d3c8521ba6f65738bf8a172690efd8d4fef35b3de9f2976ed8b66675cf5",
            "track_name": "api-cpu-agents1-no-preload",
            "measurement_hash": "c7d7714aeb30a156e468b742747141d2b0c0c834db8621f004f7705067ccf84c",
            "eval_algorithm": "shared-agent-no-preload@1",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[7200.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        },
        {
          "entry_id": 145,
          "entry_name": "GPT-6 Astra · medium · svc_94_2d132c",
          "model": "GPT-6 Astra",
          "effort": "medium",
          "effort_rank": 2,
          "model_type": "closed",
          "performance": 0.8834210526315789,
          "time_per_task_sec": 495.36015863834217,
          "average_time_per_task_sec": 495.36015863834217,
          "median_time_per_task_sec": 419.82664201750003,
          "cost_usd": 4.332102315789474,
          "turns_per_task": null,
          "release_date": "2026-09-03",
          "average_output_tokens": null,
          "frontier": false,
          "qualification_known": false,
          "qualifying": false,
          "reference": false,
          "provisional": false,
          "run_id": "svc_94_2d132c",
          "source_archive_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/447ff4cd77230ee06ee5b0e864873c43ceb0138c/pranjal-gpt6astra-codex-mypcbench-energy38-medium-100steps-no-preload-modal-v1-94.zip",
          "cost_basis": "published_API_rate_estimate_from_logged_usage_and_cache_not_recorded_charge",
          "notes": "Published HF PR25, merged before inventory. GPT-6 Astra via Codex; 100 CUA batches, 7200-second timeout, no-preload track, 38 original partial-credit scores retained. Cost is an API-equivalent estimate from complete logged usage and cache counts at the published September 12 reference rates, not Codex-plan spending or a current-price quote. 3/38 judge bundles lacked supplied final-response text. 2 agent(s) exited 1 after the gateway finalized at the 100-batch limit; recorded verdicts unchanged. One large paste action was omitted from the judge bundle for preference_inference_f025; full action remains in runlog, and the original score of 43 is retained.",
          "quality_flags": "{\"Astra_cost_excludes_per_request_long_context_fast_mode_and_tool_adjustments\":38,\"agent_exit_1_after_gateway_finalized_at_100_batches\":2,\"codex_model_response_count_and_full_tool_calls_unavailable\":38,\"judge_did_not_receive_final_response_text\":3,\"publisher_reports_one_large_action_omitted_from_judge_bundle_full_runlog_preserved\":1}",
          "selected_task_seed_set_sha256": "ef48e19fb8e187616e5b3c2afa7b4bc1d27a8a3d18c3efe0c48148c668a7b35e",
          "run_terminal_events": "[\"run_done\"]",
          "row_type": "evaluation",
          "source_cost_usd": 4.332102315789474,
          "cost_kind": "source_estimate",
          "cost_label": "API estimate · logged usage",
          "cost_note": "published_API_rate_estimate_from_logged_usage_and_cache_not_recorded_charge. not an invoice; excludes unlogged requests, startup credential probes, infrastructure and separate grading charges",
          "cost_range_min_usd": null,
          "cost_range_max_usd": null,
          "cost_coverage_tasks": 38,
          "cost_total_tasks": 38,
          "cost_source": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/447ff4cd77230ee06ee5b0e864873c43ceb0138c/COST-pranjal-eval137.json (reference rates recorded 2026-09-12T05:10:02.711841+00:00; input/cache/write/output=10/1/12.5/50 USD per million; not current-price verification)",
          "cost_source_url": "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/447ff4cd77230ee06ee5b0e864873c43ceb0138c/COST-pranjal-eval137.json",
          "cost_pricing_model": "",
          "cost_unit_prices": null,
          "model_metadata": {
            "model_type": "closed",
            "release_date": "2026-09-03",
            "release_source": "https://developers.openai.com/api/docs/changelog",
            "access_source": "https://developers.openai.com/api/docs/models/gpt-6-astra",
            "notes": "OpenAI's September 3, 2026 changelog announces GPT-6 Astra. Proprietary hosted model.",
            "model": "GPT-6 Astra",
            "verified_on": "2026-09-16"
          },
          "source_run_ids": [
            "svc_94_2d132c"
          ],
          "source_archive_urls": [
            "https://huggingface.co/datasets/anonymousmypcbench/cua-speedrun-trajectories/resolve/447ff4cd77230ee06ee5b0e864873c43ceb0138c/pranjal-gpt6astra-codex-mypcbench-energy38-medium-100steps-no-preload-modal-v1-94.zip"
          ],
          "task_count": 38,
          "steps_per_task": 36.71052631578947,
          "tool_calls_per_task": 97.39473684210526,
          "model_responses_per_task": null,
          "variant": "Codex",
          "series": "GPT-6 Astra · Codex",
          "source_metadata": {
            "split_dset_name": "mypcbench-energy38",
            "benchmark_name": "mypcbench-energy38",
            "benchmark_version": "0.2",
            "template_name": "codex_cli",
            "benchmark_content_hash": "77ba0d3c8521ba6f65738bf8a172690efd8d4fef35b3de9f2976ed8b66675cf5",
            "track_name": "api-cpu-agents1-no-preload",
            "measurement_hash": "c7d7714aeb30a156e468b742747141d2b0c0c834db8621f004f7705067ccf84c",
            "eval_algorithm": "shared-agent-no-preload@1",
            "agents_per_evaluation": "1",
            "timeout_sec_values": "[7200.0]",
            "inference_tier": "not_recorded",
            "io_variant": "as_archived",
            "backend": "modal-native",
            "time_definition": "environment-owned task clock; agent plus environment; excludes setup and final verification",
            "parallel_evaluations": "8"
          }
        }
      ],
      "excluded_evaluations": [],
      "supporting_rows": 0,
      "task_counts": [
        38
      ],
      "paper_subset": {
        "label": "MyPCBench",
        "selected_tasks": 38,
        "eligible_tasks": 184
      }
    }
  },
  "excluded": {
    "Older reference (2026-09-03); retaining newer medium run svc_61_1ec4be (2026-09-13, no preload). Not a failed run.": 1,
    "Rate-limited playground contrast; contributor-tier high run svc_58_38520a retained.": 1,
    "Rate-limited playground contrast; contributor-tier low run svc_60_72500b retained.": 1,
    "Rate-limited playground contrast; contributor-tier medium run svc_59_d8a6d4 retained.": 1,
    "Rate-limited playground contrast; contributor-tier minimal run svc_61_96ce3f retained.": 1,
    "Rate-limited playground contrast; contributor-tier xhigh run svc_69_41f88a retained.": 1,
    "billing-affected": 1,
    "combined source": 2,
    "retained_extra_attempts": 35,
    "run-failed": 3,
    "superseded configuration label": 1,
    "superseded snapshot": 2
  },
  "source_rows": 146,
  "model_metadata_verified_on": "2026-09-19",
  "curation_evidence": "docs/static-results-datasets.md#reviewed-variants-and-exclusions",
  "included_rows": 85,
  "inventory_utc": "2026-09-20T20:08:33Z",
  "publication_scope": {
    "datasets": [
      "osworld-energy50-representative",
      "osworld2-k52",
      "cua-world-long-k26",
      "mypcbench-energy38"
    ],
    "omitted_reviewed_rows": 11,
    "note": "Paper task subsets. The source CSV retains the full archive."
  }
}
