{
  "schema": 1,
  "generated_at": "2026-09-02T05:35:13+00:00",
  "benchmark": {
    "id": "integration-bench",
    "name": "Integration Bench",
    "rev": "rev 01",
    "status": "preliminary",
    "sweep": "50 public tasks, 1 official attempt per model, Aug 2026",
    "scorer": "task-score-v4-mandatory-gated",
    "metric": "Task Score (0-100, mean over 50 tasks)"
  },
  "counts": {
    "models": 17,
    "tasks": 10,
    "tasks_scored": 50,
    "attempts": 170,
    "attempts_scored": 850,
    "vendors": 6
  },
  "surfaces": [
    {
      "id": "polling",
      "n": 42,
      "published": 8,
      "task_score_mean": 58.62
    },
    {
      "id": "writeback",
      "n": 21,
      "published": 6,
      "task_score_mean": 59.08
    },
    {
      "id": "webhooks",
      "n": 11,
      "published": 2,
      "task_score_mean": 19.63
    }
  ],
  "models": [
    {
      "model": "claude-fable-5",
      "name": "Claude Fable 5",
      "vendor": "anthropic",
      "harness": [
        "claude-code"
      ],
      "harness_label": "Claude Code",
      "reasoning_effort": [
        "xhigh"
      ],
      "status": "scored",
      "rank": 1
    },
    {
      "model": "gpt-5.6-sol",
      "name": "GPT-5.6 Sol",
      "vendor": "openai",
      "harness": [
        "codex"
      ],
      "harness_label": "Codex CLI",
      "reasoning_effort": [
        "xhigh"
      ],
      "status": "scored",
      "rank": 2
    },
    {
      "model": "deepseek-v4-pro-max",
      "name": "DeepSeek V4 Pro (max)",
      "vendor": "deepseek",
      "harness": [
        "opencode"
      ],
      "harness_label": "OpenCode",
      "reasoning_effort": [
        "max"
      ],
      "status": "scored",
      "rank": 3
    },
    {
      "model": "qwen-3.8-2.4t-a95b",
      "name": "Qwen 3.8 2.4T A95B",
      "vendor": "qwen",
      "harness": [
        "opencode"
      ],
      "harness_label": "OpenCode",
      "reasoning_effort": [
        "unverified-provider-default"
      ],
      "status": "scored",
      "rank": 4
    },
    {
      "model": "grok-4.6",
      "name": "Grok 4.6",
      "vendor": "grok",
      "harness": [
        "opencode"
      ],
      "harness_label": "OpenCode",
      "reasoning_effort": [
        "unverified-provider-default"
      ],
      "status": "scored",
      "rank": 5
    },
    {
      "model": "claude-opus-5",
      "name": "Claude Opus 5",
      "vendor": "anthropic",
      "harness": [
        "claude-code"
      ],
      "harness_label": "Claude Code",
      "reasoning_effort": [
        "xhigh"
      ],
      "status": "scored",
      "rank": 6
    },
    {
      "model": "qwen-3.8-27b",
      "name": "Qwen 3.8 27B (xhigh)",
      "vendor": "qwen",
      "harness": [
        "opencode"
      ],
      "harness_label": "OpenCode",
      "reasoning_effort": [
        "xhigh"
      ],
      "status": "scored",
      "rank": 7
    },
    {
      "model": "kimi-k3",
      "name": "Kimi K3",
      "vendor": "moonshot",
      "harness": [
        "opencode"
      ],
      "harness_label": "OpenCode",
      "reasoning_effort": [
        "unverified-provider-default"
      ],
      "status": "scored",
      "rank": 8
    },
    {
      "model": "deepseek-v4-flash-max",
      "name": "DeepSeek V4 Flash (max)",
      "vendor": "deepseek",
      "harness": [
        "opencode"
      ],
      "harness_label": "OpenCode",
      "reasoning_effort": [
        "max"
      ],
      "status": "scored",
      "rank": 9
    },
    {
      "model": "claude-opus-4-8",
      "name": "Claude Opus 4.8",
      "vendor": "anthropic",
      "harness": [
        "claude-code"
      ],
      "harness_label": "Claude Code",
      "reasoning_effort": [
        "xhigh"
      ],
      "status": "scored",
      "rank": 10
    },
    {
      "model": "claude-sonnet-5",
      "name": "Claude Sonnet 5",
      "vendor": "anthropic",
      "harness": [
        "claude-code"
      ],
      "harness_label": "Claude Code",
      "reasoning_effort": [
        "xhigh"
      ],
      "status": "scored",
      "rank": 11
    },
    {
      "model": "gpt-5.6-terra",
      "name": "GPT-5.6 Terra",
      "vendor": "openai",
      "harness": [
        "codex"
      ],
      "harness_label": "Codex CLI",
      "reasoning_effort": [
        "xhigh"
      ],
      "status": "scored",
      "rank": 12
    },
    {
      "model": "muse-spark-1.2",
      "name": "Muse Spark 1.2",
      "vendor": "meta",
      "harness": [
        "opencode"
      ],
      "harness_label": "OpenCode",
      "reasoning_effort": [
        "unverified-provider-default"
      ],
      "status": "scored",
      "rank": 13
    },
    {
      "model": "gpt-5.6-luna",
      "name": "GPT-5.6 Luna",
      "vendor": "openai",
      "harness": [
        "codex"
      ],
      "harness_label": "Codex CLI",
      "reasoning_effort": [
        "xhigh"
      ],
      "status": "scored",
      "rank": 14
    },
    {
      "model": "glm-5.2",
      "name": "GLM 5.2 (xhigh)",
      "vendor": "zhipu",
      "harness": [
        "opencode"
      ],
      "harness_label": "OpenCode",
      "reasoning_effort": [
        "xhigh"
      ],
      "status": "scored",
      "rank": 15
    },
    {
      "model": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "vendor": "anthropic",
      "harness": [
        "claude-code"
      ],
      "harness_label": "Claude Code",
      "reasoning_effort": [
        "xhigh"
      ],
      "status": "scored",
      "rank": 16
    },
    {
      "model": "gemini-3.7-flash",
      "name": "Gemini 3.7 Flash",
      "vendor": "gemini",
      "harness": [
        "opencode"
      ],
      "harness_label": "OpenCode",
      "reasoning_effort": [
        "unverified-provider-default"
      ],
      "status": "incomplete",
      "rank": 17
    }
  ],
  "tasks": [
    {
      "id": "task-0001",
      "title": "Sync StaffLine candidates, jobs, applications, and notes into our canonical store",
      "vendor": "StaffLine",
      "surface": "polling (pull)",
      "surfaces": [
        "polling"
      ],
      "models_attempted": 17,
      "resolved_by": 14,
      "task_score_mean": 77.94,
      "task_score_max": 100.0,
      "checks": 30
    },
    {
      "id": "task-0011",
      "title": "TalentForge connector: consume webhooks while pushing corrections upstream",
      "vendor": null,
      "surface": "webhooks (push) and writeback (POST/PATCH), across both vendors",
      "surfaces": [
        "writeback",
        "webhooks"
      ],
      "models_attempted": 17,
      "resolved_by": 0,
      "task_score_mean": 0.0,
      "task_score_max": 0.0,
      "checks": 52
    },
    {
      "id": "task-0015",
      "title": "Bulk-import a migration batch into StaffLine",
      "vendor": "StaffLine",
      "surface": "writeback (bulk create)",
      "surfaces": [
        "writeback"
      ],
      "models_attempted": 17,
      "resolved_by": 1,
      "task_score_mean": 5.88,
      "task_score_max": 100.0,
      "checks": 26
    },
    {
      "id": "task-0016",
      "title": "HireWire connector: push stage-change events + keep an incremental poll fresh",
      "vendor": "HireWire",
      "surface": "writeback (PATCH + POST) and polling (incremental read)",
      "surfaces": [
        "polling",
        "writeback"
      ],
      "models_attempted": 17,
      "resolved_by": 17,
      "task_score_mean": 93.24,
      "task_score_max": 100.0,
      "checks": 48
    },
    {
      "id": "task-0020",
      "title": "Sync GlobalHire candidates into our canonical store (offset polling)",
      "vendor": "GlobalHire",
      "surface": "polling (pull)",
      "surfaces": [
        "polling"
      ],
      "models_attempted": 17,
      "resolved_by": 12,
      "task_score_mean": 70.59,
      "task_score_max": 100.0,
      "checks": 34
    },
    {
      "id": "task-0025",
      "title": "Reset the CrewCall roster watermark after the tenant rebuild",
      "vendor": "CrewCall",
      "surface": "polling",
      "surfaces": [
        "polling"
      ],
      "models_attempted": 17,
      "resolved_by": 0,
      "task_score_mean": 0.0,
      "task_score_max": 0.0,
      "checks": 503
    },
    {
      "id": "task-0037",
      "title": "Fix the Placemint redeployment sync",
      "vendor": "Placemint",
      "surface": "polling, writeback",
      "surfaces": [
        "polling",
        "writeback"
      ],
      "models_attempted": 17,
      "resolved_by": 14,
      "task_score_mean": 82.35,
      "task_score_max": 100.0,
      "checks": 306
    },
    {
      "id": "task-0042",
      "title": "Move the rota mirror onto instant storage",
      "vendor": "Rosterly",
      "surface": "polling",
      "surfaces": [
        "polling"
      ],
      "models_attempted": 17,
      "resolved_by": 12,
      "task_score_mean": 70.59,
      "task_score_max": 100.0,
      "checks": 117
    },
    {
      "id": "task-0047",
      "title": "Reconcile and safely apply the mobility stage repair queue",
      "vendor": "GlobalHire",
      "surface": "record reads and batch writeback",
      "surfaces": [
        "polling",
        "writeback"
      ],
      "models_attempted": 17,
      "resolved_by": 8,
      "task_score_mean": 47.06,
      "task_score_max": 100.0,
      "checks": 100
    },
    {
      "id": "task-0049",
      "title": "Placemint connector: webhooks, polling, and writeback for a high-volume tenant",
      "vendor": "Placemint",
      "surface": "webhooks (push), polling (pull), writeback (push)",
      "surfaces": [
        "polling",
        "writeback",
        "webhooks"
      ],
      "models_attempted": 17,
      "resolved_by": 16,
      "task_score_mean": 93.64,
      "task_score_max": 100.0,
      "checks": 69
    }
  ],
  "paths": {
    "leaderboard": "leaderboard.json",
    "matrix": "matrix.json",
    "model": "models/{model}.json",
    "task": "tasks/{task}.json",
    "attempt": "attempts/{model}/{task}/"
  },
  "caveats": [
    "One official attempt per model per task; run-to-run variance is not characterised.",
    "Scores use task-score-v4 (mandatory-gated): a failed mandatory check zeroes the task.",
    "Leaderboard means are over all 50 tasks of the public split; per-task results and trajectories are published for 10 of them (initial release).",
    "gemini-3.7-flash: 22 of 50 tasks hit provider failures and are scored 0 in the published mean.",
    "Surfaces (polling/writeback/webhooks) come from the ticket's Surface line and are non-exclusive."
  ]
}
