{
  "artifact_version": "1",
  "report_id": "support-urgency-v1-support-urgency-1791554979029",
  "generated_at": "2026-10-09T14:55:23.800Z",
  "instruction_hash": "sha256:ea691d7e6d27119ed8eeaf4beff48523b4bfd4337c7675990fb653b78bb29012",
  "harness_version": "support-urgency-benchmark-v1",
  "runner_commit": "57982be10f9150a69326544426cc965fc4f7d5de",
  "dataset": {
    "id": "support-urgency",
    "version": "v1",
    "hash": "sha256:decc1c2d9ac55655e2f959b4585dc414a3e706f5e5ea627efe00dfbe0de7e24a",
    "label_policy_hash": "sha256:416b2735ea0b246ecdc427e96d52bfc9cd1ea594a29360c132456a69069d17f3",
    "license": {
      "id": "CC-BY-4.0",
      "name": "Creative Commons Attribution 4.0 International",
      "url": "https://creativecommons.org/licenses/by/4.0/"
    },
    "sample_count": {
      "development": 200,
      "test": 600
    }
  },
  "headline": {
    "id": "errors_at_default_and_own_threshold",
    "rationale": "Every decision model ranks urgent messages above the rest almost perfectly, so the comparison is where each model’s probabilities sit: errors at the 0.5 threshold, errors at a threshold chosen on the development cases, and what another model’s threshold costs."
  },
  "run": {
    "safety": {
      "environment_id": "runner-v1",
      "region": "unspecified"
    },
    "started_at": "2026-10-09T14:09:39.067Z",
    "finished_at": "2026-10-09T14:55:23.800Z",
    "max_calls": 4000,
    "call_totals": {
      "planned": 4000,
      "issued": 4000,
      "completed": 4000,
      "succeeded": 3998,
      "failed": 2
    },
    "approved_spend_ceiling_usd": "5.60",
    "spend_ceiling_enforced_by_runner": true,
    "run_ids": {
      "development": [
        "support-urgency-1791554979029"
      ],
      "test": [
        "support-urgency-1791556067681"
      ]
    }
  },
  "methodology": {
    "same_frozen_cases": true,
    "same_instructions": true,
    "concurrency": 1,
    "retry_policy": "none",
    "cache": "disabled",
    "failed_attempts_remain_in_denominators": true,
    "yes_threshold": 0.5,
    "thresholds_chosen_on": "development",
    "results_measured_on": "test"
  },
  "models": [
    {
      "provider": "typesafe",
      "model_id": "jev",
      "requested_revision": "jev-latest",
      "resolved_revisions": [
        "jev-1.13.0"
      ],
      "attempted": 800,
      "valid": 800
    },
    {
      "provider": "cloudflare",
      "model_id": "clef",
      "requested_revision": "clef",
      "resolved_revisions": [
        "clef"
      ],
      "attempted": 800,
      "valid": 798
    },
    {
      "provider": "cloudflare",
      "model_id": "clef-flash",
      "requested_revision": "clef-flash",
      "resolved_revisions": [
        "clef-flash"
      ],
      "attempted": 800,
      "valid": 800
    },
    {
      "provider": "openai",
      "model_id": "openai-decisions",
      "requested_revision": "gpt-6-luna",
      "resolved_revisions": [
        "gpt-6-luna"
      ],
      "attempted": 800,
      "valid": 800
    },
    {
      "provider": "openai",
      "model_id": "gpt-6-luna",
      "requested_revision": "gpt-6-luna",
      "resolved_revisions": [
        "gpt-6-luna"
      ],
      "attempted": 800,
      "valid": 800
    }
  ],
  "columns": [
    "dataset_id",
    "dataset_version",
    "dataset_hash",
    "run_id",
    "case_id",
    "split",
    "state",
    "expected",
    "kind",
    "category",
    "difficulty",
    "provider",
    "model_id",
    "requested_revision",
    "resolved_revision",
    "status",
    "error_class",
    "answer",
    "probability_yes",
    "started_at",
    "finished_at",
    "latency_ms",
    "input_tokens",
    "output_tokens"
  ]
}
