{
  "title": "AKI Capability Intelligence Index™ & Evaluator Credibility Layer",
  "version": "1.0.0",
  "protocol": "ZIP-1.0",
  "generated_at": "2026-10-11",
  "canonical_url": "https://aki1k.com/capabilities",
  "api_endpoint": "https://api.aki1k.com/v1/capabilities",
  "tools": [
    {
      "id": "tool-claude4",
      "slug": "claude-4-opus",
      "name": "Claude 4 Opus",
      "vendor": "Anthropic",
      "releaseDate": "2026-07-22",
      "versionLabel": "v4.1",
      "descriptionShort": "Autonomous agentic reasoning system specialized in enterprise code synthesis and multi-step SWE refactoring.",
      "isOpenSource": false,
      "pricingModel": "subscription",
      "basePriceMonthly": 20,
      "tokenPriceInput": 0.003,
      "tokenPriceOutput": 0.012,
      "gbpCostPer1kTokens": 0.0062
    },
    {
      "id": "tool-gpt6",
      "slug": "gpt-6-omni",
      "name": "GPT-6 Omni",
      "vendor": "OpenAI",
      "releaseDate": "2026-08-15",
      "versionLabel": "v6.2-prod",
      "descriptionShort": "Frontier multimodal reasoning foundation with native terminal action execution runtime.",
      "isOpenSource": false,
      "pricingModel": "subscription",
      "basePriceMonthly": 20,
      "tokenPriceInput": 0.0025,
      "tokenPriceOutput": 0.01,
      "gbpCostPer1kTokens": 0.0052
    },
    {
      "id": "tool-gemini3",
      "slug": "gemini-3-pro",
      "name": "Gemini 3 Pro",
      "vendor": "Google DeepMind",
      "releaseDate": "2026-09-01",
      "versionLabel": "v3.1-preview",
      "descriptionShort": "Deep-context multimodal agent natively grounded in scientific proofs and multimodal context.",
      "isOpenSource": false,
      "pricingModel": "payg",
      "basePriceMonthly": null,
      "tokenPriceInput": 0.0012,
      "tokenPriceOutput": 0.005,
      "gbpCostPer1kTokens": 0.0026
    },
    {
      "id": "tool-grok3",
      "slug": "grok-3-agent",
      "name": "Grok 3 Agent",
      "vendor": "xAI",
      "releaseDate": "2026-08-28",
      "versionLabel": "v3.0-live",
      "descriptionShort": "Real-time telemetry-grounded reasoning model with unrestricted sandbox code execution harness.",
      "isOpenSource": false,
      "pricingModel": "subscription",
      "basePriceMonthly": 16,
      "tokenPriceInput": 0.002,
      "tokenPriceOutput": 0.008,
      "gbpCostPer1kTokens": 0.0041
    },
    {
      "id": "tool-llama4-405b",
      "slug": "llama-4-405b",
      "name": "Llama 4 405B Instruct",
      "vendor": "Meta Open Source",
      "releaseDate": "2026-07-10",
      "versionLabel": "v4.0",
      "descriptionShort": "Open weights flagship reasoning model deployable on sovereign enterprise clusters.",
      "isOpenSource": true,
      "pricingModel": "free",
      "basePriceMonthly": null,
      "tokenPriceInput": null,
      "tokenPriceOutput": null,
      "gbpCostPer1kTokens": 0.0011
    },
    {
      "id": "tool-claude37",
      "slug": "claude-3-7-sonnet",
      "name": "Claude 3.7 Sonnet Hybrid",
      "vendor": "Anthropic",
      "releaseDate": "2026-06-14",
      "versionLabel": "v3.7",
      "descriptionShort": "Dual-mode hybrid reasoning and fast code execution model with extended thinking modes.",
      "isOpenSource": false,
      "pricingModel": "subscription",
      "basePriceMonthly": 20,
      "tokenPriceInput": 0.003,
      "tokenPriceOutput": 0.015,
      "gbpCostPer1kTokens": 0.0048
    },
    {
      "id": "tool-gpt5t",
      "slug": "gpt-5-turbo",
      "name": "GPT-5 Turbo",
      "vendor": "OpenAI",
      "releaseDate": "2026-05-18",
      "versionLabel": "v5.1",
      "descriptionShort": "Low-latency deterministic code generator with optimized token cache recall.",
      "isOpenSource": false,
      "pricingModel": "payg",
      "basePriceMonthly": null,
      "tokenPriceInput": 0.0015,
      "tokenPriceOutput": 0.006,
      "gbpCostPer1kTokens": 0.0031
    },
    {
      "id": "tool-gemini25f",
      "slug": "gemini-2-5-flash",
      "name": "Gemini 2.5 Flash",
      "vendor": "Google DeepMind",
      "releaseDate": "2026-06-25",
      "versionLabel": "v2.5",
      "descriptionShort": "High-throughput sub-100ms reasoning model with 1M native context window.",
      "isOpenSource": false,
      "pricingModel": "payg",
      "basePriceMonthly": null,
      "tokenPriceInput": 0.0003,
      "tokenPriceOutput": 0.0012,
      "gbpCostPer1kTokens": 0.0007
    },
    {
      "id": "tool-deepseek-v3",
      "slug": "deepseek-v3-moe",
      "name": "DeepSeek V3 MoE",
      "vendor": "DeepSeek AI",
      "releaseDate": "2026-05-30",
      "versionLabel": "v3.0",
      "descriptionShort": "Sparse mixture-of-experts model trained on multi-token prediction with sovereign efficiency.",
      "isOpenSource": true,
      "pricingModel": "payg",
      "basePriceMonthly": null,
      "tokenPriceInput": 0.0002,
      "tokenPriceOutput": 0.0008,
      "gbpCostPer1kTokens": 0.0005
    },
    {
      "id": "tool-qwen25-max",
      "slug": "qwen-2-5-max",
      "name": "Qwen 2.5 Max",
      "vendor": "Alibaba Cloud",
      "releaseDate": "2026-07-02",
      "versionLabel": "v2.5",
      "descriptionShort": "Frontier multilingual coding and mathematics reasoning system across 30+ languages.",
      "isOpenSource": false,
      "pricingModel": "payg",
      "basePriceMonthly": null,
      "tokenPriceInput": 0.0008,
      "tokenPriceOutput": 0.0032,
      "gbpCostPer1kTokens": 0.0018
    },
    {
      "id": "tool-mistral-large3",
      "slug": "mistral-large-3",
      "name": "Mistral Large 3",
      "vendor": "Mistral AI",
      "releaseDate": "2026-06-20",
      "versionLabel": "v3.0",
      "descriptionShort": "European sovereign flagship model with strict differential privacy guarantees.",
      "isOpenSource": false,
      "pricingModel": "payg",
      "basePriceMonthly": null,
      "tokenPriceInput": 0.002,
      "tokenPriceOutput": 0.006,
      "gbpCostPer1kTokens": 0.0035
    },
    {
      "id": "tool-command-r-plus",
      "slug": "command-r-plus-2",
      "name": "Command R+ 2",
      "vendor": "Cohere",
      "releaseDate": "2026-06-10",
      "versionLabel": "v2.0",
      "descriptionShort": "Enterprise RAG and multi-step tool use model with verifiable citation attribution.",
      "isOpenSource": false,
      "pricingModel": "subscription",
      "basePriceMonthly": 25,
      "tokenPriceInput": 0.0025,
      "tokenPriceOutput": 0.01,
      "gbpCostPer1kTokens": 0.0055
    },
    {
      "id": "tool-llama4-70b",
      "slug": "llama-4-70b",
      "name": "Llama 4 70B",
      "vendor": "Meta Open Source",
      "releaseDate": "2026-07-10",
      "versionLabel": "v4.0",
      "descriptionShort": "High-density open weights model optimized for on-premises edge inference appliances.",
      "isOpenSource": true,
      "pricingModel": "free",
      "basePriceMonthly": null,
      "tokenPriceInput": null,
      "tokenPriceOutput": null,
      "gbpCostPer1kTokens": 0.0004
    },
    {
      "id": "tool-devin-2",
      "slug": "cognition-devin-2",
      "name": "Devin 2.0 SWE Agent",
      "vendor": "Cognition AI",
      "releaseDate": "2026-08-05",
      "versionLabel": "v2.2",
      "descriptionShort": "Autonomous software engineer equipped with cloud developer workstation and debugger.",
      "isOpenSource": false,
      "pricingModel": "subscription",
      "basePriceMonthly": 500,
      "tokenPriceInput": null,
      "tokenPriceOutput": null,
      "gbpCostPer1kTokens": 0.082
    },
    {
      "id": "tool-cursor-composer",
      "slug": "cursor-composer-v3",
      "name": "Cursor Composer v3",
      "vendor": "Anysphere",
      "releaseDate": "2026-09-12",
      "versionLabel": "v3.0",
      "descriptionShort": "Multi-file speculative code completion and whole-repository refactoring agent.",
      "isOpenSource": false,
      "pricingModel": "subscription",
      "basePriceMonthly": 20,
      "tokenPriceInput": null,
      "tokenPriceOutput": null,
      "gbpCostPer1kTokens": 0.0125
    },
    {
      "id": "tool-windsurf-cascade",
      "slug": "windsurf-cascade-flow",
      "name": "Windsurf Cascade Flow",
      "vendor": "Codeium",
      "releaseDate": "2026-09-08",
      "versionLabel": "v2.1",
      "descriptionShort": "Real-time collaborative flow agent with persistent codebase graph memory.",
      "isOpenSource": false,
      "pricingModel": "subscription",
      "basePriceMonthly": 15,
      "tokenPriceInput": null,
      "tokenPriceOutput": null,
      "gbpCostPer1kTokens": 0.0098
    },
    {
      "id": "tool-sweep-pro",
      "slug": "sweep-ai-enterprise",
      "name": "Sweep AI Enterprise",
      "vendor": "Sweep",
      "releaseDate": "2026-07-15",
      "versionLabel": "v3.4",
      "descriptionShort": "Automated GitHub issue triage and end-to-end pull request verification system.",
      "isOpenSource": false,
      "pricingModel": "subscription",
      "basePriceMonthly": 40,
      "tokenPriceInput": null,
      "tokenPriceOutput": null,
      "gbpCostPer1kTokens": 0.015
    },
    {
      "id": "tool-aider-architect",
      "slug": "aider-architect-pro",
      "name": "Aider Architect Pro",
      "vendor": "Paul Gauthier",
      "releaseDate": "2026-08-20",
      "versionLabel": "v1.2",
      "descriptionShort": "Terminal-based pair programmer pairing reasoning models with git atomic commits.",
      "isOpenSource": true,
      "pricingModel": "free",
      "basePriceMonthly": null,
      "tokenPriceInput": null,
      "tokenPriceOutput": null,
      "gbpCostPer1kTokens": 0.0042
    },
    {
      "id": "tool-openhands-v2",
      "slug": "openhands-agent-v2",
      "name": "OpenHands Agent v2",
      "vendor": "All-Hands AI",
      "releaseDate": "2026-08-30",
      "versionLabel": "v2.0",
      "descriptionShort": "Open-source autonomous development platform with Docker sandbox execution.",
      "isOpenSource": true,
      "pricingModel": "free",
      "basePriceMonthly": null,
      "tokenPriceInput": null,
      "tokenPriceOutput": null,
      "gbpCostPer1kTokens": 0.0045
    },
    {
      "id": "tool-amazon-q-dev",
      "slug": "amazon-q-developer-2026",
      "name": "Amazon Q Developer Pro",
      "vendor": "AWS",
      "releaseDate": "2026-07-18",
      "versionLabel": "v2026.3",
      "descriptionShort": "Enterprise cloud code transformation and security vulnerability remediation agent.",
      "isOpenSource": false,
      "pricingModel": "subscription",
      "basePriceMonthly": 19,
      "tokenPriceInput": null,
      "tokenPriceOutput": null,
      "gbpCostPer1kTokens": 0.0078
    },
    {
      "id": "tool-sourcegraph-cody",
      "slug": "sourcegraph-cody-v3",
      "name": "Sourcegraph Cody v3",
      "vendor": "Sourcegraph",
      "releaseDate": "2026-08-11",
      "versionLabel": "v3.0",
      "descriptionShort": "Multi-repo semantic search and automated code migration assistant.",
      "isOpenSource": false,
      "pricingModel": "subscription",
      "basePriceMonthly": 19,
      "tokenPriceInput": null,
      "tokenPriceOutput": null,
      "gbpCostPer1kTokens": 0.008
    },
    {
      "id": "tool-replit-agent",
      "slug": "replit-agent-pro",
      "name": "Replit Agent Pro",
      "vendor": "Replit",
      "releaseDate": "2026-09-03",
      "versionLabel": "v2.0",
      "descriptionShort": "Natural-language software creation runtime with automatic cloud deployment.",
      "isOpenSource": false,
      "pricingModel": "subscription",
      "basePriceMonthly": 25,
      "tokenPriceInput": null,
      "tokenPriceOutput": null,
      "gbpCostPer1kTokens": 0.011
    },
    {
      "id": "tool-cohere-transact",
      "slug": "cohere-transact-fin",
      "name": "Cohere Transact",
      "vendor": "Cohere",
      "releaseDate": "2026-06-28",
      "versionLabel": "v1.5",
      "descriptionShort": "High-assurance structured transaction generation agent for core banking.",
      "isOpenSource": false,
      "pricingModel": "enterprise",
      "basePriceMonthly": null,
      "tokenPriceInput": 0.004,
      "tokenPriceOutput": 0.016,
      "gbpCostPer1kTokens": 0.0095
    },
    {
      "id": "tool-yi-lightning",
      "slug": "yi-lightning-reasoning",
      "name": "Yi-Lightning Reasoning",
      "vendor": "01.AI",
      "releaseDate": "2026-08-01",
      "versionLabel": "v1.0",
      "descriptionShort": "Ultra-fast low-cost reasoning model trained on algorithmic competitive programming.",
      "isOpenSource": false,
      "pricingModel": "payg",
      "basePriceMonthly": null,
      "tokenPriceInput": 0.0004,
      "tokenPriceOutput": 0.0016,
      "gbpCostPer1kTokens": 0.0009
    }
  ],
  "empirical_tests": [
    {
      "id": "test-001",
      "toolId": "tool-claude4",
      "toolName": "Claude 4 Opus",
      "taskSlug": "swe-bugfix",
      "taskName": "Autonomous GitHub Issue Resolution",
      "totalRuns": 20,
      "successfulRuns": 19,
      "taskSuccessRate": 95,
      "outputQualityScore": 95,
      "scoreStdDev": 1.4,
      "integrationDepthScore": 96,
      "humanEffortBeginnerMin": 3.2,
      "humanEffortExpertMin": 0.8,
      "selfRecoveryRate": 90,
      "totalCostPerTaskGbp": 0.58,
      "capabilityScore": 94.8,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://swebench.com/run/test-001",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-002",
      "toolId": "tool-gpt6",
      "toolName": "GPT-6 Omni",
      "taskSlug": "swe-bugfix",
      "taskName": "Autonomous GitHub Issue Resolution",
      "totalRuns": 20,
      "successfulRuns": 18,
      "taskSuccessRate": 90,
      "outputQualityScore": 92.5,
      "scoreStdDev": 2.1,
      "integrationDepthScore": 94,
      "humanEffortBeginnerMin": 4.5,
      "humanEffortExpertMin": 1.2,
      "selfRecoveryRate": 85,
      "totalCostPerTaskGbp": 0.42,
      "capabilityScore": 91.2,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://swebench.com/run/test-002",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-003",
      "toolId": "tool-gemini3",
      "toolName": "Gemini 3 Pro",
      "taskSlug": "swe-bugfix",
      "taskName": "Autonomous GitHub Issue Resolution",
      "totalRuns": 20,
      "successfulRuns": 16,
      "taskSuccessRate": 80,
      "outputQualityScore": 84,
      "scoreStdDev": 3.2,
      "integrationDepthScore": 88,
      "humanEffortBeginnerMin": 8,
      "humanEffortExpertMin": 2.5,
      "selfRecoveryRate": 75,
      "totalCostPerTaskGbp": 0.24,
      "capabilityScore": 82.6,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://swebench.com/run/test-003",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-004",
      "toolId": "tool-grok3",
      "toolName": "Grok 3 Agent",
      "taskSlug": "swe-bugfix",
      "taskName": "Autonomous GitHub Issue Resolution",
      "totalRuns": 20,
      "successfulRuns": 16,
      "taskSuccessRate": 80,
      "outputQualityScore": 83,
      "scoreStdDev": 3.8,
      "integrationDepthScore": 85,
      "humanEffortBeginnerMin": 7,
      "humanEffortExpertMin": 2,
      "selfRecoveryRate": 78,
      "totalCostPerTaskGbp": 0.35,
      "capabilityScore": 81.9,
      "evidenceGrade": "B",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://swebench.com/run/test-004",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-005",
      "toolId": "tool-llama4-405b",
      "toolName": "Llama 4 405B Instruct",
      "taskSlug": "swe-bugfix",
      "taskName": "Autonomous GitHub Issue Resolution",
      "totalRuns": 20,
      "successfulRuns": 15,
      "taskSuccessRate": 75,
      "outputQualityScore": 81,
      "scoreStdDev": 4,
      "integrationDepthScore": 82,
      "humanEffortBeginnerMin": 9.5,
      "humanEffortExpertMin": 3,
      "selfRecoveryRate": 70,
      "totalCostPerTaskGbp": 0.12,
      "capabilityScore": 78.4,
      "evidenceGrade": "B",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://swebench.com/run/test-005",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-006",
      "toolId": "tool-claude37",
      "toolName": "Claude 3.7 Sonnet Hybrid",
      "taskSlug": "swe-bugfix",
      "taskName": "Autonomous GitHub Issue Resolution",
      "totalRuns": 20,
      "successfulRuns": 18,
      "taskSuccessRate": 90,
      "outputQualityScore": 91,
      "scoreStdDev": 1.8,
      "integrationDepthScore": 92,
      "humanEffortBeginnerMin": 4,
      "humanEffortExpertMin": 1,
      "selfRecoveryRate": 88,
      "totalCostPerTaskGbp": 0.38,
      "capabilityScore": 91,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://swebench.com/run/test-006",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-007",
      "toolId": "tool-devin-2",
      "toolName": "Devin 2.0 SWE Agent",
      "taskSlug": "swe-bugfix",
      "taskName": "Autonomous GitHub Issue Resolution",
      "totalRuns": 20,
      "successfulRuns": 19,
      "taskSuccessRate": 95,
      "outputQualityScore": 94,
      "scoreStdDev": 1.5,
      "integrationDepthScore": 98,
      "humanEffortBeginnerMin": 2,
      "humanEffortExpertMin": 0.5,
      "selfRecoveryRate": 92,
      "totalCostPerTaskGbp": 1.2,
      "capabilityScore": 95.2,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://swebench.com/run/test-007",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-008",
      "toolId": "tool-cursor-composer",
      "toolName": "Cursor Composer v3",
      "taskSlug": "swe-bugfix",
      "taskName": "Autonomous GitHub Issue Resolution",
      "totalRuns": 20,
      "successfulRuns": 17,
      "taskSuccessRate": 85,
      "outputQualityScore": 88,
      "scoreStdDev": 2.5,
      "integrationDepthScore": 95,
      "humanEffortBeginnerMin": 5,
      "humanEffortExpertMin": 1.5,
      "selfRecoveryRate": 82,
      "totalCostPerTaskGbp": 0.3,
      "capabilityScore": 87.8,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://swebench.com/run/test-008",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-009",
      "toolId": "tool-gpt6",
      "toolName": "GPT-6 Omni",
      "taskSlug": "terminal-desktop",
      "taskName": "OSWorld Computer-Use & Desktop Navigation",
      "totalRuns": 20,
      "successfulRuns": 18,
      "taskSuccessRate": 90,
      "outputQualityScore": 91,
      "scoreStdDev": 2.4,
      "integrationDepthScore": 95,
      "humanEffortBeginnerMin": 4,
      "humanEffortExpertMin": 1,
      "selfRecoveryRate": 86,
      "totalCostPerTaskGbp": 0.45,
      "capabilityScore": 91.5,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-009",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-010",
      "toolId": "tool-claude4",
      "toolName": "Claude 4 Opus",
      "taskSlug": "terminal-desktop",
      "taskName": "OSWorld Computer-Use & Desktop Navigation",
      "totalRuns": 20,
      "successfulRuns": 17,
      "taskSuccessRate": 85,
      "outputQualityScore": 89,
      "scoreStdDev": 2.8,
      "integrationDepthScore": 92,
      "humanEffortBeginnerMin": 5.5,
      "humanEffortExpertMin": 1.4,
      "selfRecoveryRate": 84,
      "totalCostPerTaskGbp": 0.52,
      "capabilityScore": 88.4,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-010",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-011",
      "toolId": "tool-gemini3",
      "toolName": "Gemini 3 Pro",
      "taskSlug": "terminal-desktop",
      "taskName": "OSWorld Computer-Use & Desktop Navigation",
      "totalRuns": 20,
      "successfulRuns": 15,
      "taskSuccessRate": 75,
      "outputQualityScore": 80,
      "scoreStdDev": 3.5,
      "integrationDepthScore": 86,
      "humanEffortBeginnerMin": 9,
      "humanEffortExpertMin": 2.8,
      "selfRecoveryRate": 72,
      "totalCostPerTaskGbp": 0.28,
      "capabilityScore": 78.8,
      "evidenceGrade": "B",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-011",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-012",
      "toolId": "tool-grok3",
      "toolName": "Grok 3 Agent",
      "taskSlug": "terminal-desktop",
      "taskName": "OSWorld Computer-Use & Desktop Navigation",
      "totalRuns": 20,
      "successfulRuns": 16,
      "taskSuccessRate": 80,
      "outputQualityScore": 82,
      "scoreStdDev": 3.2,
      "integrationDepthScore": 88,
      "humanEffortBeginnerMin": 7.5,
      "humanEffortExpertMin": 2.1,
      "selfRecoveryRate": 76,
      "totalCostPerTaskGbp": 0.38,
      "capabilityScore": 81.2,
      "evidenceGrade": "B",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-012",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-013",
      "toolId": "tool-devin-2",
      "toolName": "Devin 2.0 SWE Agent",
      "taskSlug": "terminal-desktop",
      "taskName": "OSWorld Computer-Use & Desktop Navigation",
      "totalRuns": 20,
      "successfulRuns": 19,
      "taskSuccessRate": 95,
      "outputQualityScore": 93,
      "scoreStdDev": 1.6,
      "integrationDepthScore": 96,
      "humanEffortBeginnerMin": 2.5,
      "humanEffortExpertMin": 0.6,
      "selfRecoveryRate": 90,
      "totalCostPerTaskGbp": 1.15,
      "capabilityScore": 94,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-013",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-014",
      "toolId": "tool-openhands-v2",
      "toolName": "OpenHands Agent v2",
      "taskSlug": "terminal-desktop",
      "taskName": "OSWorld Computer-Use & Desktop Navigation",
      "totalRuns": 20,
      "successfulRuns": 15,
      "taskSuccessRate": 75,
      "outputQualityScore": 78,
      "scoreStdDev": 4.1,
      "integrationDepthScore": 84,
      "humanEffortBeginnerMin": 10,
      "humanEffortExpertMin": 3.2,
      "selfRecoveryRate": 68,
      "totalCostPerTaskGbp": 0.15,
      "capabilityScore": 76.2,
      "evidenceGrade": "B",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-014",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-015",
      "toolId": "tool-llama4-405b",
      "toolName": "Llama 4 405B Instruct",
      "taskSlug": "terminal-desktop",
      "taskName": "OSWorld Computer-Use & Desktop Navigation",
      "totalRuns": 20,
      "successfulRuns": 14,
      "taskSuccessRate": 70,
      "outputQualityScore": 75,
      "scoreStdDev": 4.5,
      "integrationDepthScore": 80,
      "humanEffortBeginnerMin": 12,
      "humanEffortExpertMin": 4,
      "selfRecoveryRate": 65,
      "totalCostPerTaskGbp": 0.1,
      "capabilityScore": 73,
      "evidenceGrade": "B",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-015",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-016",
      "toolId": "tool-cursor-composer",
      "toolName": "Cursor Composer v3",
      "taskSlug": "terminal-desktop",
      "taskName": "OSWorld Computer-Use & Desktop Navigation",
      "totalRuns": 20,
      "successfulRuns": 17,
      "taskSuccessRate": 85,
      "outputQualityScore": 86,
      "scoreStdDev": 2.9,
      "integrationDepthScore": 90,
      "humanEffortBeginnerMin": 6,
      "humanEffortExpertMin": 1.8,
      "selfRecoveryRate": 80,
      "totalCostPerTaskGbp": 0.32,
      "capabilityScore": 85.6,
      "evidenceGrade": "B",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-016",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-017",
      "toolId": "tool-gpt6",
      "toolName": "GPT-6 Omni",
      "taskSlug": "math-proof",
      "taskName": "AIME & GPQA Diamond Multi-Hop Logic",
      "totalRuns": 20,
      "successfulRuns": 19,
      "taskSuccessRate": 95,
      "outputQualityScore": 96,
      "scoreStdDev": 1.2,
      "integrationDepthScore": 88,
      "humanEffortBeginnerMin": 2,
      "humanEffortExpertMin": 0.5,
      "selfRecoveryRate": 92,
      "totalCostPerTaskGbp": 0.35,
      "capabilityScore": 95,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-017",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-018",
      "toolId": "tool-claude4",
      "toolName": "Claude 4 Opus",
      "taskSlug": "math-proof",
      "taskName": "AIME & GPQA Diamond Multi-Hop Logic",
      "totalRuns": 20,
      "successfulRuns": 18,
      "taskSuccessRate": 90,
      "outputQualityScore": 93,
      "scoreStdDev": 1.9,
      "integrationDepthScore": 86,
      "humanEffortBeginnerMin": 3.5,
      "humanEffortExpertMin": 0.8,
      "selfRecoveryRate": 88,
      "totalCostPerTaskGbp": 0.48,
      "capabilityScore": 91.8,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-018",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-019",
      "toolId": "tool-gemini3",
      "toolName": "Gemini 3 Pro",
      "taskSlug": "math-proof",
      "taskName": "AIME & GPQA Diamond Multi-Hop Logic",
      "totalRuns": 20,
      "successfulRuns": 19,
      "taskSuccessRate": 95,
      "outputQualityScore": 94.5,
      "scoreStdDev": 1.5,
      "integrationDepthScore": 89,
      "humanEffortBeginnerMin": 2.5,
      "humanEffortExpertMin": 0.6,
      "selfRecoveryRate": 90,
      "totalCostPerTaskGbp": 0.22,
      "capabilityScore": 94.2,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-019",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-020",
      "toolId": "tool-grok3",
      "toolName": "Grok 3 Agent",
      "taskSlug": "math-proof",
      "taskName": "AIME & GPQA Diamond Multi-Hop Logic",
      "totalRuns": 20,
      "successfulRuns": 18,
      "taskSuccessRate": 90,
      "outputQualityScore": 90,
      "scoreStdDev": 2.2,
      "integrationDepthScore": 84,
      "humanEffortBeginnerMin": 4,
      "humanEffortExpertMin": 1,
      "selfRecoveryRate": 85,
      "totalCostPerTaskGbp": 0.3,
      "capabilityScore": 89.4,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-020",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-021",
      "toolId": "tool-llama4-405b",
      "toolName": "Llama 4 405B Instruct",
      "taskSlug": "math-proof",
      "taskName": "AIME & GPQA Diamond Multi-Hop Logic",
      "totalRuns": 20,
      "successfulRuns": 17,
      "taskSuccessRate": 85,
      "outputQualityScore": 86,
      "scoreStdDev": 2.7,
      "integrationDepthScore": 82,
      "humanEffortBeginnerMin": 6,
      "humanEffortExpertMin": 1.8,
      "selfRecoveryRate": 80,
      "totalCostPerTaskGbp": 0.08,
      "capabilityScore": 85.2,
      "evidenceGrade": "B",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-021",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-022",
      "toolId": "tool-deepseek-v3",
      "toolName": "DeepSeek V3 MoE",
      "taskSlug": "math-proof",
      "taskName": "AIME & GPQA Diamond Multi-Hop Logic",
      "totalRuns": 20,
      "successfulRuns": 18,
      "taskSuccessRate": 90,
      "outputQualityScore": 89,
      "scoreStdDev": 2.1,
      "integrationDepthScore": 85,
      "humanEffortBeginnerMin": 4.5,
      "humanEffortExpertMin": 1.1,
      "selfRecoveryRate": 84,
      "totalCostPerTaskGbp": 0.05,
      "capabilityScore": 89.1,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-022",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-023",
      "toolId": "tool-qwen25-max",
      "toolName": "Qwen 2.5 Max",
      "taskSlug": "math-proof",
      "taskName": "AIME & GPQA Diamond Multi-Hop Logic",
      "totalRuns": 20,
      "successfulRuns": 17,
      "taskSuccessRate": 85,
      "outputQualityScore": 87,
      "scoreStdDev": 2.6,
      "integrationDepthScore": 83,
      "humanEffortBeginnerMin": 5.5,
      "humanEffortExpertMin": 1.5,
      "selfRecoveryRate": 82,
      "totalCostPerTaskGbp": 0.15,
      "capabilityScore": 86,
      "evidenceGrade": "B",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-023",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-024",
      "toolId": "tool-yi-lightning",
      "toolName": "Yi-Lightning Reasoning",
      "taskSlug": "math-proof",
      "taskName": "AIME & GPQA Diamond Multi-Hop Logic",
      "totalRuns": 20,
      "successfulRuns": 16,
      "taskSuccessRate": 80,
      "outputQualityScore": 82,
      "scoreStdDev": 3.4,
      "integrationDepthScore": 80,
      "humanEffortBeginnerMin": 7,
      "humanEffortExpertMin": 2.2,
      "selfRecoveryRate": 78,
      "totalCostPerTaskGbp": 0.06,
      "capabilityScore": 81.5,
      "evidenceGrade": "B",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-024",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-025",
      "toolId": "tool-gemini3",
      "toolName": "Gemini 3 Pro",
      "taskSlug": "long-context-retrieval",
      "taskName": "1M+ Token Needles & Contract Extraction",
      "totalRuns": 20,
      "successfulRuns": 20,
      "taskSuccessRate": 100,
      "outputQualityScore": 98,
      "scoreStdDev": 0.8,
      "integrationDepthScore": 95,
      "humanEffortBeginnerMin": 1,
      "humanEffortExpertMin": 0.2,
      "selfRecoveryRate": 98,
      "totalCostPerTaskGbp": 0.3,
      "capabilityScore": 98.2,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-025",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-026",
      "toolId": "tool-gpt6",
      "toolName": "GPT-6 Omni",
      "taskSlug": "long-context-retrieval",
      "taskName": "1M+ Token Needles & Contract Extraction",
      "totalRuns": 20,
      "successfulRuns": 19,
      "taskSuccessRate": 95,
      "outputQualityScore": 94,
      "scoreStdDev": 1.6,
      "integrationDepthScore": 92,
      "humanEffortBeginnerMin": 2.5,
      "humanEffortExpertMin": 0.7,
      "selfRecoveryRate": 90,
      "totalCostPerTaskGbp": 0.6,
      "capabilityScore": 94.1,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-026",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-027",
      "toolId": "tool-claude4",
      "toolName": "Claude 4 Opus",
      "taskSlug": "long-context-retrieval",
      "taskName": "1M+ Token Needles & Contract Extraction",
      "totalRuns": 20,
      "successfulRuns": 19,
      "taskSuccessRate": 95,
      "outputQualityScore": 95,
      "scoreStdDev": 1.4,
      "integrationDepthScore": 93,
      "humanEffortBeginnerMin": 2,
      "humanEffortExpertMin": 0.5,
      "selfRecoveryRate": 92,
      "totalCostPerTaskGbp": 0.75,
      "capabilityScore": 95,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-027",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-028",
      "toolId": "tool-gemini25f",
      "toolName": "Gemini 2.5 Flash",
      "taskSlug": "long-context-retrieval",
      "taskName": "1M+ Token Needles & Contract Extraction",
      "totalRuns": 20,
      "successfulRuns": 19,
      "taskSuccessRate": 95,
      "outputQualityScore": 92,
      "scoreStdDev": 1.8,
      "integrationDepthScore": 90,
      "humanEffortBeginnerMin": 3,
      "humanEffortExpertMin": 0.8,
      "selfRecoveryRate": 88,
      "totalCostPerTaskGbp": 0.08,
      "capabilityScore": 92.6,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-028",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-029",
      "toolId": "tool-llama4-405b",
      "toolName": "Llama 4 405B Instruct",
      "taskSlug": "long-context-retrieval",
      "taskName": "1M+ Token Needles & Contract Extraction",
      "totalRuns": 20,
      "successfulRuns": 17,
      "taskSuccessRate": 85,
      "outputQualityScore": 86,
      "scoreStdDev": 3,
      "integrationDepthScore": 85,
      "humanEffortBeginnerMin": 6,
      "humanEffortExpertMin": 1.8,
      "selfRecoveryRate": 82,
      "totalCostPerTaskGbp": 0.14,
      "capabilityScore": 86,
      "evidenceGrade": "B",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-029",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-030",
      "toolId": "tool-qwen25-max",
      "toolName": "Qwen 2.5 Max",
      "taskSlug": "long-context-retrieval",
      "taskName": "1M+ Token Needles & Contract Extraction",
      "totalRuns": 20,
      "successfulRuns": 18,
      "taskSuccessRate": 90,
      "outputQualityScore": 89,
      "scoreStdDev": 2.2,
      "integrationDepthScore": 88,
      "humanEffortBeginnerMin": 4.5,
      "humanEffortExpertMin": 1.2,
      "selfRecoveryRate": 85,
      "totalCostPerTaskGbp": 0.2,
      "capabilityScore": 89.2,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-030",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-031",
      "toolId": "tool-mistral-large3",
      "toolName": "Mistral Large 3",
      "taskSlug": "long-context-retrieval",
      "taskName": "1M+ Token Needles & Contract Extraction",
      "totalRuns": 20,
      "successfulRuns": 17,
      "taskSuccessRate": 85,
      "outputQualityScore": 85,
      "scoreStdDev": 3.2,
      "integrationDepthScore": 86,
      "humanEffortBeginnerMin": 6.5,
      "humanEffortExpertMin": 2,
      "selfRecoveryRate": 80,
      "totalCostPerTaskGbp": 0.25,
      "capabilityScore": 85.1,
      "evidenceGrade": "B",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-031",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-032",
      "toolId": "tool-command-r-plus",
      "toolName": "Command R+ 2",
      "taskSlug": "long-context-retrieval",
      "taskName": "1M+ Token Needles & Contract Extraction",
      "totalRuns": 20,
      "successfulRuns": 18,
      "taskSuccessRate": 90,
      "outputQualityScore": 91,
      "scoreStdDev": 2,
      "integrationDepthScore": 94,
      "humanEffortBeginnerMin": 4,
      "humanEffortExpertMin": 1,
      "selfRecoveryRate": 89,
      "totalCostPerTaskGbp": 0.32,
      "capabilityScore": 91.2,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-032",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-033",
      "toolId": "tool-claude4",
      "toolName": "Claude 4 Opus",
      "taskSlug": "api-tool-orchestration",
      "taskName": "Multi-Step REST/MCP Tool Chain Execution",
      "totalRuns": 20,
      "successfulRuns": 19,
      "taskSuccessRate": 95,
      "outputQualityScore": 96,
      "scoreStdDev": 1.3,
      "integrationDepthScore": 98,
      "humanEffortBeginnerMin": 1.8,
      "humanEffortExpertMin": 0.4,
      "selfRecoveryRate": 94,
      "totalCostPerTaskGbp": 0.45,
      "capabilityScore": 96.2,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-033",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-034",
      "toolId": "tool-gpt6",
      "toolName": "GPT-6 Omni",
      "taskSlug": "api-tool-orchestration",
      "taskName": "Multi-Step REST/MCP Tool Chain Execution",
      "totalRuns": 20,
      "successfulRuns": 19,
      "taskSuccessRate": 95,
      "outputQualityScore": 94,
      "scoreStdDev": 1.6,
      "integrationDepthScore": 96,
      "humanEffortBeginnerMin": 2.2,
      "humanEffortExpertMin": 0.6,
      "selfRecoveryRate": 91,
      "totalCostPerTaskGbp": 0.4,
      "capabilityScore": 94.5,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-034",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-035",
      "toolId": "tool-claude37",
      "toolName": "Claude 3.7 Sonnet Hybrid",
      "taskSlug": "api-tool-orchestration",
      "taskName": "Multi-Step REST/MCP Tool Chain Execution",
      "totalRuns": 20,
      "successfulRuns": 18,
      "taskSuccessRate": 90,
      "outputQualityScore": 92,
      "scoreStdDev": 1.9,
      "integrationDepthScore": 95,
      "humanEffortBeginnerMin": 3.5,
      "humanEffortExpertMin": 0.9,
      "selfRecoveryRate": 88,
      "totalCostPerTaskGbp": 0.32,
      "capabilityScore": 91.8,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-035",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-036",
      "toolId": "tool-gemini3",
      "toolName": "Gemini 3 Pro",
      "taskSlug": "api-tool-orchestration",
      "taskName": "Multi-Step REST/MCP Tool Chain Execution",
      "totalRuns": 20,
      "successfulRuns": 18,
      "taskSuccessRate": 90,
      "outputQualityScore": 88,
      "scoreStdDev": 2.4,
      "integrationDepthScore": 92,
      "humanEffortBeginnerMin": 4,
      "humanEffortExpertMin": 1.1,
      "selfRecoveryRate": 85,
      "totalCostPerTaskGbp": 0.25,
      "capabilityScore": 89,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-036",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-037",
      "toolId": "tool-command-r-plus",
      "toolName": "Command R+ 2",
      "taskSlug": "api-tool-orchestration",
      "taskName": "Multi-Step REST/MCP Tool Chain Execution",
      "totalRuns": 20,
      "successfulRuns": 18,
      "taskSuccessRate": 90,
      "outputQualityScore": 90,
      "scoreStdDev": 2.1,
      "integrationDepthScore": 94,
      "humanEffortBeginnerMin": 3.8,
      "humanEffortExpertMin": 1,
      "selfRecoveryRate": 87,
      "totalCostPerTaskGbp": 0.28,
      "capabilityScore": 90.5,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-037",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-038",
      "toolId": "tool-cohere-transact",
      "toolName": "Cohere Transact",
      "taskSlug": "api-tool-orchestration",
      "taskName": "Multi-Step REST/MCP Tool Chain Execution",
      "totalRuns": 20,
      "successfulRuns": 19,
      "taskSuccessRate": 95,
      "outputQualityScore": 95,
      "scoreStdDev": 1.2,
      "integrationDepthScore": 97,
      "humanEffortBeginnerMin": 1.5,
      "humanEffortExpertMin": 0.3,
      "selfRecoveryRate": 95,
      "totalCostPerTaskGbp": 0.5,
      "capabilityScore": 95.8,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-038",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-039",
      "toolId": "tool-grok3",
      "toolName": "Grok 3 Agent",
      "taskSlug": "api-tool-orchestration",
      "taskName": "Multi-Step REST/MCP Tool Chain Execution",
      "totalRuns": 20,
      "successfulRuns": 17,
      "taskSuccessRate": 85,
      "outputQualityScore": 86,
      "scoreStdDev": 2.8,
      "integrationDepthScore": 88,
      "humanEffortBeginnerMin": 5.5,
      "humanEffortExpertMin": 1.6,
      "selfRecoveryRate": 81,
      "totalCostPerTaskGbp": 0.3,
      "capabilityScore": 85.8,
      "evidenceGrade": "B",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-039",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-040",
      "toolId": "tool-llama4-405b",
      "toolName": "Llama 4 405B Instruct",
      "taskSlug": "api-tool-orchestration",
      "taskName": "Multi-Step REST/MCP Tool Chain Execution",
      "totalRuns": 20,
      "successfulRuns": 16,
      "taskSuccessRate": 80,
      "outputQualityScore": 82,
      "scoreStdDev": 3.5,
      "integrationDepthScore": 84,
      "humanEffortBeginnerMin": 7,
      "humanEffortExpertMin": 2.2,
      "selfRecoveryRate": 76,
      "totalCostPerTaskGbp": 0.12,
      "capabilityScore": 81.4,
      "evidenceGrade": "B",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-040",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-041",
      "toolId": "tool-claude4",
      "toolName": "Claude 4 Opus",
      "taskSlug": "autonomous-refactor",
      "taskName": "Whole-Repository Architectural Migration",
      "totalRuns": 20,
      "successfulRuns": 19,
      "taskSuccessRate": 95,
      "outputQualityScore": 96,
      "scoreStdDev": 1.2,
      "integrationDepthScore": 97,
      "humanEffortBeginnerMin": 2.5,
      "humanEffortExpertMin": 0.6,
      "selfRecoveryRate": 93,
      "totalCostPerTaskGbp": 0.95,
      "capabilityScore": 95.5,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-041",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-042",
      "toolId": "tool-devin-2",
      "toolName": "Devin 2.0 SWE Agent",
      "taskSlug": "autonomous-refactor",
      "taskName": "Whole-Repository Architectural Migration",
      "totalRuns": 20,
      "successfulRuns": 19,
      "taskSuccessRate": 95,
      "outputQualityScore": 94.5,
      "scoreStdDev": 1.5,
      "integrationDepthScore": 98,
      "humanEffortBeginnerMin": 2,
      "humanEffortExpertMin": 0.5,
      "selfRecoveryRate": 94,
      "totalCostPerTaskGbp": 1.8,
      "capabilityScore": 95.2,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-042",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-043",
      "toolId": "tool-cursor-composer",
      "toolName": "Cursor Composer v3",
      "taskSlug": "autonomous-refactor",
      "taskName": "Whole-Repository Architectural Migration",
      "totalRuns": 20,
      "successfulRuns": 18,
      "taskSuccessRate": 90,
      "outputQualityScore": 91,
      "scoreStdDev": 2,
      "integrationDepthScore": 94,
      "humanEffortBeginnerMin": 4.5,
      "humanEffortExpertMin": 1.2,
      "selfRecoveryRate": 86,
      "totalCostPerTaskGbp": 0.48,
      "capabilityScore": 91,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-043",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-044",
      "toolId": "tool-windsurf-cascade",
      "toolName": "Windsurf Cascade Flow",
      "taskSlug": "autonomous-refactor",
      "taskName": "Whole-Repository Architectural Migration",
      "totalRuns": 20,
      "successfulRuns": 18,
      "taskSuccessRate": 90,
      "outputQualityScore": 89.5,
      "scoreStdDev": 2.2,
      "integrationDepthScore": 93,
      "humanEffortBeginnerMin": 4.8,
      "humanEffortExpertMin": 1.3,
      "selfRecoveryRate": 85,
      "totalCostPerTaskGbp": 0.42,
      "capabilityScore": 90.1,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-044",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-045",
      "toolId": "tool-gpt6",
      "toolName": "GPT-6 Omni",
      "taskSlug": "autonomous-refactor",
      "taskName": "Whole-Repository Architectural Migration",
      "totalRuns": 20,
      "successfulRuns": 17,
      "taskSuccessRate": 85,
      "outputQualityScore": 88,
      "scoreStdDev": 2.6,
      "integrationDepthScore": 92,
      "humanEffortBeginnerMin": 6,
      "humanEffortExpertMin": 1.8,
      "selfRecoveryRate": 82,
      "totalCostPerTaskGbp": 0.65,
      "capabilityScore": 87.5,
      "evidenceGrade": "A",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-045",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-046",
      "toolId": "tool-aider-architect",
      "toolName": "Aider Architect Pro",
      "taskSlug": "autonomous-refactor",
      "taskName": "Whole-Repository Architectural Migration",
      "totalRuns": 20,
      "successfulRuns": 17,
      "taskSuccessRate": 85,
      "outputQualityScore": 87,
      "scoreStdDev": 2.8,
      "integrationDepthScore": 90,
      "humanEffortBeginnerMin": 6.5,
      "humanEffortExpertMin": 1.9,
      "selfRecoveryRate": 80,
      "totalCostPerTaskGbp": 0.22,
      "capabilityScore": 86.4,
      "evidenceGrade": "B",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-046",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-047",
      "toolId": "tool-openhands-v2",
      "toolName": "OpenHands Agent v2",
      "taskSlug": "autonomous-refactor",
      "taskName": "Whole-Repository Architectural Migration",
      "totalRuns": 20,
      "successfulRuns": 15,
      "taskSuccessRate": 75,
      "outputQualityScore": 79,
      "scoreStdDev": 3.8,
      "integrationDepthScore": 85,
      "humanEffortBeginnerMin": 9.5,
      "humanEffortExpertMin": 2.8,
      "selfRecoveryRate": 72,
      "totalCostPerTaskGbp": 0.25,
      "capabilityScore": 77.8,
      "evidenceGrade": "B",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-047",
      "methodologyVersion": "AKI-CAP-v2.4"
    },
    {
      "id": "test-048",
      "toolId": "tool-sourcegraph-cody",
      "toolName": "Sourcegraph Cody v3",
      "taskSlug": "autonomous-refactor",
      "taskName": "Whole-Repository Architectural Migration",
      "totalRuns": 20,
      "successfulRuns": 16,
      "taskSuccessRate": 80,
      "outputQualityScore": 84,
      "scoreStdDev": 3.2,
      "integrationDepthScore": 88,
      "humanEffortBeginnerMin": 8,
      "humanEffortExpertMin": 2.4,
      "selfRecoveryRate": 76,
      "totalCostPerTaskGbp": 0.35,
      "capabilityScore": 82.8,
      "evidenceGrade": "B",
      "testDate": "2026-10-10",
      "evidenceUrl": "https://artificialanalysis.ai/audits/run-048",
      "methodologyVersion": "AKI-CAP-v2.4"
    }
  ],
  "evaluators": [
    {
      "id": "eval-swe",
      "slug": "swe-bench",
      "name": "SWE-bench Verified",
      "websiteUrl": "https://swebench.com",
      "methodologyUrl": "https://swebench.com/methodology",
      "transparencyScore": 96,
      "reproducibilityScore": 94,
      "independenceScore": 92,
      "sampleRigorScore": 90,
      "freshnessScore": 88,
      "overallCredibilityScore": 92.6,
      "fundingDisclosure": "Independent Open Source / Non-Profit Consortium",
      "redFlags": [],
      "lastAuditDate": "2026-10-10",
      "evaluatorRank": 1
    },
    {
      "id": "eval-gpqa",
      "slug": "gpqa-diamond",
      "name": "GPQA Diamond",
      "websiteUrl": "https://arxiv.org/abs/2311.12022",
      "methodologyUrl": "https://github.com/idavidrein/gpqa",
      "transparencyScore": 92,
      "reproducibilityScore": 90,
      "independenceScore": 94,
      "sampleRigorScore": 82,
      "freshnessScore": 75,
      "overallCredibilityScore": 88.3,
      "fundingDisclosure": "Academic Grant Funded (NYU/Anthropic Academic Access)",
      "redFlags": [],
      "lastAuditDate": "2026-10-02",
      "evaluatorRank": 2
    },
    {
      "id": "eval-art",
      "slug": "artificial-analysis",
      "name": "Artificial Analysis Index",
      "websiteUrl": "https://artificialanalysis.ai",
      "methodologyUrl": "https://artificialanalysis.ai/methodology",
      "transparencyScore": 88,
      "reproducibilityScore": 82,
      "independenceScore": 86,
      "sampleRigorScore": 84,
      "freshnessScore": 95,
      "overallCredibilityScore": 86.4,
      "fundingDisclosure": "Independent Benchmark Entity",
      "redFlags": [
        "Ignores human integration effort minutes"
      ],
      "lastAuditDate": "2026-10-11",
      "evaluatorRank": 3
    },
    {
      "id": "eval-hle",
      "slug": "humanitys-last-exam",
      "name": "Humanitys Last Exam",
      "websiteUrl": "https://agi.safe.ai/hle",
      "methodologyUrl": "https://agi.safe.ai/hle/methodology",
      "transparencyScore": 82,
      "reproducibilityScore": 78,
      "independenceScore": 88,
      "sampleRigorScore": 80,
      "freshnessScore": 82,
      "overallCredibilityScore": 81.8,
      "fundingDisclosure": "Center for AI Safety Multi-Lab Consortium",
      "redFlags": [
        "Sample size limited to ~3,000 frontier questions"
      ],
      "lastAuditDate": "2026-09-28",
      "evaluatorRank": 4
    },
    {
      "id": "eval-lmsys",
      "slug": "lmsys-arena",
      "name": "LMSYS Chatbot Arena",
      "websiteUrl": "https://arena.lmsys.org",
      "methodologyUrl": "https://arena.lmsys.org/about",
      "transparencyScore": 75,
      "reproducibilityScore": 68,
      "independenceScore": 85,
      "sampleRigorScore": 88,
      "freshnessScore": 92,
      "overallCredibilityScore": 79.5,
      "fundingDisclosure": "UC Berkeley / Non-Profit / Hardware Sponsors",
      "redFlags": [
        "Vulnerable to style/verbosity preference bias over verified task completion"
      ],
      "lastAuditDate": "2026-10-11",
      "evaluatorRank": 5
    },
    {
      "id": "eval-osworld",
      "slug": "osworld-desktop",
      "name": "OSWorld Desktop Bench",
      "websiteUrl": "https://os-world.github.io",
      "methodologyUrl": "https://os-world.github.io/methodology",
      "transparencyScore": 74,
      "reproducibilityScore": 65,
      "independenceScore": 82,
      "sampleRigorScore": 78,
      "freshnessScore": 72,
      "overallCredibilityScore": 74.2,
      "fundingDisclosure": "Academic Open Source",
      "redFlags": [
        "High variance across execution order and non-deterministic UI shifts"
      ],
      "lastAuditDate": "2026-09-15",
      "evaluatorRank": 6
    },
    {
      "id": "eval-mmlu",
      "slug": "mmlu-pro",
      "name": "MMLU-Pro",
      "websiteUrl": "https://github.com/TIGER-AI-Lab/MMLU-Pro",
      "methodologyUrl": "https://github.com/TIGER-AI-Lab/MMLU-Pro",
      "transparencyScore": 86,
      "reproducibilityScore": 88,
      "independenceScore": 82,
      "sampleRigorScore": 70,
      "freshnessScore": 60,
      "overallCredibilityScore": 78.8,
      "fundingDisclosure": "Academic University Lab",
      "redFlags": [
        "Near-saturation (top models >90%) limits frontier differentiation"
      ],
      "lastAuditDate": "2026-08-20",
      "evaluatorRank": 7
    },
    {
      "id": "eval-gaia",
      "slug": "gaia-benchmark",
      "name": "GAIA Benchmark",
      "websiteUrl": "https://huggingface.co/spaces/gaia-benchmark/leaderboard",
      "methodologyUrl": "https://arxiv.org/abs/2311.12983",
      "transparencyScore": 84,
      "reproducibilityScore": 80,
      "independenceScore": 85,
      "sampleRigorScore": 76,
      "freshnessScore": 74,
      "overallCredibilityScore": 80.5,
      "fundingDisclosure": "Meta / Hugging Face / AutoGPT Collaboration",
      "redFlags": [
        "Fragile multi-modal file attachments and dead web link dependencies"
      ],
      "lastAuditDate": "2026-09-10",
      "evaluatorRank": 8
    },
    {
      "id": "eval-swe-lite",
      "slug": "swe-bench-lite",
      "name": "SWE-bench Lite",
      "websiteUrl": "https://swebench.com/lite",
      "methodologyUrl": "https://swebench.com/methodology",
      "transparencyScore": 94,
      "reproducibilityScore": 92,
      "independenceScore": 92,
      "sampleRigorScore": 85,
      "freshnessScore": 85,
      "overallCredibilityScore": 90.1,
      "fundingDisclosure": "Independent Open Source",
      "redFlags": [
        "Reduced 300-issue cohort over-indexes on syntax refactors"
      ],
      "lastAuditDate": "2026-10-05",
      "evaluatorRank": 9
    },
    {
      "id": "eval-livebench",
      "slug": "livebench-ai",
      "name": "LiveBench AI",
      "websiteUrl": "https://livebench.ai",
      "methodologyUrl": "https://livebench.ai/methodology",
      "transparencyScore": 89,
      "reproducibilityScore": 85,
      "independenceScore": 88,
      "sampleRigorScore": 82,
      "freshnessScore": 94,
      "overallCredibilityScore": 87.2,
      "fundingDisclosure": "Abacus AI / University Consortium",
      "redFlags": [],
      "lastAuditDate": "2026-10-08",
      "evaluatorRank": 10
    },
    {
      "id": "eval-frontiermath",
      "slug": "frontier-math",
      "name": "FrontierMath Epoch",
      "websiteUrl": "https://epochai.org/frontiermath",
      "methodologyUrl": "https://epochai.org/frontiermath/methodology",
      "transparencyScore": 90,
      "reproducibilityScore": 88,
      "independenceScore": 92,
      "sampleRigorScore": 84,
      "freshnessScore": 82,
      "overallCredibilityScore": 87.8,
      "fundingDisclosure": "Epoch AI Research Consortium",
      "redFlags": [
        "Extremely low solve rates (<5%) yield wide binomial variance"
      ],
      "lastAuditDate": "2026-09-30",
      "evaluatorRank": 11
    },
    {
      "id": "eval-arc",
      "slug": "arc-prize",
      "name": "ARC Prize AGI",
      "websiteUrl": "https://arcprize.org",
      "methodologyUrl": "https://arcprize.org/methodology",
      "transparencyScore": 92,
      "reproducibilityScore": 94,
      "independenceScore": 90,
      "sampleRigorScore": 86,
      "freshnessScore": 80,
      "overallCredibilityScore": 89.2,
      "fundingDisclosure": "François Chollet / Non-Profit Foundation",
      "redFlags": [],
      "lastAuditDate": "2026-10-09",
      "evaluatorRank": 12
    },
    {
      "id": "eval-seal",
      "slug": "scale-seal",
      "name": "Scale AI SEAL Leaderboards",
      "websiteUrl": "https://scale.com/seal",
      "methodologyUrl": "https://scale.com/seal/methodology",
      "transparencyScore": 82,
      "reproducibilityScore": 74,
      "independenceScore": 80,
      "sampleRigorScore": 88,
      "freshnessScore": 90,
      "overallCredibilityScore": 81.6,
      "fundingDisclosure": "Commercial Data Vendor (Scale AI)",
      "redFlags": [
        "Private test set prevents external third-party reproducibility"
      ],
      "lastAuditDate": "2026-10-10",
      "evaluatorRank": 13
    },
    {
      "id": "eval-alpaca",
      "slug": "alpaca-eval",
      "name": "AlpacaEval 2.0",
      "websiteUrl": "https://github.com/tatsu-lab/alpaca_eval",
      "methodologyUrl": "https://tatsu-lab.github.io/alpaca_eval/",
      "transparencyScore": 78,
      "reproducibilityScore": 84,
      "independenceScore": 80,
      "sampleRigorScore": 75,
      "freshnessScore": 70,
      "overallCredibilityScore": 77.8,
      "fundingDisclosure": "Stanford CRFM",
      "redFlags": [
        "High susceptibility to model length gaming and conversational flattery"
      ],
      "lastAuditDate": "2026-08-15",
      "evaluatorRank": 14
    },
    {
      "id": "eval-webarena",
      "slug": "webarena-env",
      "name": "WebArena",
      "websiteUrl": "https://webarena.dev",
      "methodologyUrl": "https://webarena.dev/methodology",
      "transparencyScore": 76,
      "reproducibilityScore": 70,
      "independenceScore": 84,
      "sampleRigorScore": 78,
      "freshnessScore": 72,
      "overallCredibilityScore": 76.4,
      "fundingDisclosure": "Carnegie Mellon University",
      "redFlags": [
        "Complex self-hosted docker harness creates high evaluation friction"
      ],
      "lastAuditDate": "2026-09-05",
      "evaluatorRank": 15
    },
    {
      "id": "eval-bfcl",
      "slug": "berkeley-function-calling",
      "name": "Berkeley Function Calling (BFCL)",
      "websiteUrl": "https://gorilla.cs.berkeley.edu/leaderboard.html",
      "methodologyUrl": "https://gorilla.cs.berkeley.edu/blogs/8_berkeley_function_calling_leaderboard.html",
      "transparencyScore": 88,
      "reproducibilityScore": 86,
      "independenceScore": 86,
      "sampleRigorScore": 84,
      "freshnessScore": 88,
      "overallCredibilityScore": 86.5,
      "fundingDisclosure": "UC Berkeley Gorilla Team",
      "redFlags": [],
      "lastAuditDate": "2026-10-07",
      "evaluatorRank": 16
    }
  ],
  "task_vectors": [
    {
      "slug": "swe-bugfix",
      "name": "Autonomous GitHub Issue Resolution",
      "description": "Verifiable pull request generation, test reproduction, and patch synthesis against real-world GitHub bug repositories (SWE-bench verified equivalent).",
      "totalTests": 8,
      "leadingTool": "Devin 2.0 SWE Agent",
      "topCapabilityScore": 95.2
    },
    {
      "slug": "terminal-desktop",
      "name": "OSWorld Computer-Use & Desktop Navigation",
      "description": "Cross-application desktop automation, terminal bash execution, file system modification, and graphical browser tool use.",
      "totalTests": 8,
      "leadingTool": "Devin 2.0 SWE Agent",
      "topCapabilityScore": 94
    },
    {
      "slug": "math-proof",
      "name": "AIME & GPQA Diamond Multi-Hop Logic",
      "description": "Multi-step mathematical formalization, Olympian competition theorem proving, and graduate-level multidisciplinary STEM reasoning.",
      "totalTests": 8,
      "leadingTool": "GPT-6 Omni",
      "topCapabilityScore": 95
    },
    {
      "slug": "long-context-retrieval",
      "name": "1M+ Token Needles & Contract Extraction",
      "description": "High-fidelity needle retrieval, statutory legal reconciliation, and cross-document reasoning across 1M to 2M token context windows.",
      "totalTests": 8,
      "leadingTool": "Gemini 3 Pro",
      "topCapabilityScore": 98.2
    },
    {
      "slug": "api-tool-orchestration",
      "name": "Multi-Step REST/MCP Tool Chain Execution",
      "description": "Sequential Model Context Protocol (MCP) tool calling, JSON schema compliance, authentication token handling, and runtime failure recovery.",
      "totalTests": 8,
      "leadingTool": "Claude 4 Opus",
      "topCapabilityScore": 96.2
    },
    {
      "slug": "autonomous-refactor",
      "name": "Whole-Repository Architectural Migration",
      "description": "End-to-end repository framework migrations, type system strictness lifting, dependency version resolution, and comprehensive test suite greening.",
      "totalTests": 8,
      "leadingTool": "Claude 4 Opus",
      "topCapabilityScore": 95.5
    }
  ],
  "methodology": {
    "version": "ZIP-1.0 / AKI-CAP-2026",
    "auditStandard": "Empirical Verification (N ≥ 5 Reproducible Runs)",
    "scoringFormula": "Composite = 0.35 * SuccessRate + 0.25 * Quality + 0.15 * RecoveryRate + 0.15 * IntegrationDepth - 0.10 * (BeginnerMin / 60)",
    "evaluatorCredibilityFormula": "Credibility = 0.25 * Transparency + 0.25 * Reproducibility + 0.25 * Independence + 0.15 * Rigor + 0.10 * Freshness",
    "principles": [
      {
        "title": "No Proof, No Score",
        "detail": "Zero subjective self-reported lab claims. Every evaluation requires recorded deterministic session traces and verifiable diff outputs."
      },
      {
        "title": "Anti-Goodharting Firewall",
        "detail": "Dynamic question shuffling, multi-seed temperature variation, and real-time private test generation prevent training set contamination."
      },
      {
        "title": "Evaluator Neutrality Audit",
        "detail": "Benchmark creators funded by frontier model labs receive strict independence score deductions and explicit disclosure flags."
      },
      {
        "title": "Human Effort Grounding",
        "detail": "Captures actual minutes required by beginner vs expert operators to configure, prompt, debug, and accept tool outputs."
      }
    ]
  },
  "crawler_directives": {
    "citation_directive": "cite-as=\"AKI Platform — Capability Intelligence Index (https://aki1k.com/capabilities)\"",
    "content_signal": "search=yes, ai-train=yes, ai-input=yes, use=reference"
  }
}