{
  "version": 2,
  "generatedAt": "2026-09-30T15:35:23.466Z",
  "distinctModels": 179,
  "excludedSuites": [
    {
      "suite": "SWE-bench Multimodal",
      "submissionsFound": 9,
      "reason": "The multimodal suite changed size mid-collection: 510 instances in the 2025 release (princeton-nlp/SWE-bench_Multimodal card, Jan 2025) and 480 in the current canonical one (SWE-bench/SWE-bench_Multimodal card, Aug 2026). Every multimodal submission in the experiments repo is a partial 2025-era run of 133-195 instances, so no single denominator is correct both for when the run happened and for the leaderboard a reader compares against today. Excluded rather than scored against a size that no longer applies.",
      "sources": [
        "https://huggingface.co/datasets/princeton-nlp/SWE-bench_Multimodal",
        "https://huggingface.co/datasets/SWE-bench/SWE-bench_Multimodal"
      ]
    }
  ],
  "submissionCount": 256,
  "flaggedCount": 1,
  "bySuite": {
    "Aider polyglot": 64,
    "SWE-bench Lite": 59,
    "SWE-bench Multilingual": 14,
    "SWE-bench Verified": 119
  },
  "modelLandscape": {
    "source": "https://models.dev",
    "note": "Model identity, published price, context window and capabilities. NOT benchmark data. Where several providers sell the same model, the cheapest published price is shown and providerCount records how many offer it. Prices are list prices as published by each provider and exclude caches, batch discounts and tiers.",
    "count": 2347,
    "released2026": 1412,
    "multiProvider": 600,
    "modelsFile": "/api/model-landscape.json"
  },
  "provenance": {
    "swebench": {
      "publisher": "SWE-bench project (SWE-bench org)",
      "repository": "https://github.com/SWE-bench/experiments",
      "path": "evaluation/{verified,lite,multilingual,multimodal}",
      "license": "per-submission; see each submission README"
    },
    "aider": {
      "publisher": "Aider AI (Aider-AI/aider)",
      "repository": "https://github.com/Aider-AI/aider",
      "path": "aider/website/_data/polyglot_leaderboard.yml",
      "leaderboard": "https://aider.chat/docs/leaderboards/"
    },
    "note": "All results are third-party, produced by the benchmark maintainers and contributors. CacheSphere did not run any of them and does not reproduce them. Figures are reported as published.",
    "methodNote": "SWE-bench rows: resolvedPct = instances resolved / the FULL size of the split (verified 500, lite 300, multilingual 300). The denominator is never the number of instances a submission chose to attempt — that inflates a 94-of-96 run to 97.9% when the published rate is 18.8%. Split sizes are taken never derived from a submission's own per-repo totals, which list only the repos that submission ran. `attempted` records what it actually ran, for transparency only. Aider rows: pass_rate_2 (or pass_rate_1) as published against its own exercise set. Suites and harness versions are NOT interchangeable — only compare within a suite and harness version."
  },
  "submissions": [
    {
      "id": "aider/gpt-5 (high)-2025-08-23",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-08-23",
      "harness": "diff",
      "model": "gpt-5",
      "modelSlug": "gpt-5 (high)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 88,
      "passRate": 88,
      "commitHash": "32faf82",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/o3-pro (high)-2025-06-28",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-06-28",
      "harness": "diff",
      "model": "o3-pro",
      "modelSlug": "o3-pro (high)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 84.9,
      "passRate": 84.9,
      "commitHash": "5318380",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/gemini-2.5-pro-preview-06-05 (32k think)-2025-06-06",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-06-06",
      "harness": "diff-fenced",
      "model": "gemini-2.5-pro-preview-06-05 (32k think)",
      "modelSlug": "gemini-2.5-pro-preview-06-05 (32k think)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 83.1,
      "passRate": 83.1,
      "commitHash": "f827f22",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/o3 (high)-2025-06-25",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-06-25",
      "harness": "diff",
      "model": "o3",
      "modelSlug": "o3 (high)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 81.3,
      "passRate": 81.3,
      "commitHash": "c48fea6",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/grok-4 (high)-2025-07-11",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-07-11",
      "harness": "diff",
      "model": "grok-4",
      "modelSlug": "grok-4 (high)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 79.6,
      "passRate": 79.6,
      "commitHash": "f7870b6-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/gemini-2.5-pro-preview-06-05 (default think)-2025-06-06",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-06-06",
      "harness": "diff-fenced",
      "model": "gemini-2.5-pro-preview-06-05 (default think)",
      "modelSlug": "gemini-2.5-pro-preview-06-05 (default think)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 79.1,
      "passRate": 79.1,
      "commitHash": "4c161f9-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/o3 (high) + gpt-4.1-2025-06-27",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-06-27",
      "harness": "architect",
      "model": "o3 (high) + gpt-4.1",
      "modelSlug": "o3 (high) + gpt-4.1",
      "resolved": null,
      "total": 225,
      "resolvedPct": 78.2,
      "passRate": 78.2,
      "commitHash": "4f4f00f-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/Gemini 2.5 Pro Preview 05-06-2025-05-07",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-05-07",
      "harness": "diff-fenced",
      "model": "Gemini 2.5 Pro Preview 05-06",
      "modelSlug": "Gemini 2.5 Pro Preview 05-06",
      "resolved": null,
      "total": 225,
      "resolvedPct": 76.9,
      "passRate": 76.9,
      "commitHash": "3b08327-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/DeepSeek-V3.2-Exp (Reasoner)-2025-10-03",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-10-03",
      "harness": "diff",
      "model": "DeepSeek-V3.2-Exp (Reasoner)",
      "modelSlug": "DeepSeek-V3.2-Exp (Reasoner)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 74.2,
      "passRate": 74.2,
      "commitHash": "cbb5376",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/Gemini 2.5 Pro Preview 03-25-2025-04-12",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-04-12",
      "harness": "diff-fenced",
      "model": "Gemini 2.5 Pro Preview 03-25",
      "modelSlug": "Gemini 2.5 Pro Preview 03-25",
      "resolved": null,
      "total": 225,
      "resolvedPct": 72.9,
      "passRate": 72.9,
      "commitHash": "0282574",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/o4-mini (high)-2025-04-16",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-04-16",
      "harness": "diff",
      "model": "o4-mini",
      "modelSlug": "o4-mini (high)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 72,
      "passRate": 72,
      "commitHash": "b66901f-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/claude-opus-4-20250514 (32k thinking)-2025-05-25",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-05-25",
      "harness": "diff",
      "model": "claude-opus-4-20250514 (32k thinking)",
      "modelSlug": "claude-opus-4-20250514 (32k thinking)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 72,
      "passRate": 72,
      "commitHash": "9ef3211",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/DeepSeek R1 (0528)-2025-06-06",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-06-06",
      "harness": "diff",
      "model": "DeepSeek R1 (0528)",
      "modelSlug": "DeepSeek R1 (0528)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 71.4,
      "passRate": 71.4,
      "commitHash": "4c161f9-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/claude-opus-4-20250514 (no think)-2025-05-25",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-05-25",
      "harness": "diff",
      "model": "claude-opus-4-20250514 (no think)",
      "modelSlug": "claude-opus-4-20250514 (no think)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 70.7,
      "passRate": 70.7,
      "commitHash": "9ef3211",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/DeepSeek-V3.2-Exp (Chat)-2025-10-03",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-10-03",
      "harness": "diff",
      "model": "DeepSeek-V3.2-Exp (Chat)",
      "modelSlug": "DeepSeek-V3.2-Exp (Chat)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 70.2,
      "passRate": 70.2,
      "commitHash": "cbb5376",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/claude-3-7-sonnet-20250219 (32k thinking tokens)-2025-02-24",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-02-24",
      "harness": "diff",
      "model": "claude-3-7-sonnet-20250219 (32k thinking tokens)",
      "modelSlug": "claude-3-7-sonnet-20250219 (32k thinking tokens)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 64.9,
      "passRate": 64.9,
      "commitHash": "60d11a6, 93edbda",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/DeepSeek R1 + claude-3-5-sonnet-20241022-2025-01-23",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-01-23",
      "harness": "architect",
      "model": "DeepSeek R1 + claude-3-5-sonnet-20241022",
      "modelSlug": "DeepSeek R1 + claude-3-5-sonnet-20241022",
      "resolved": null,
      "total": 225,
      "resolvedPct": 64,
      "passRate": 64,
      "commitHash": "05a77c7",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/o1-2024-12-17 (high)-2024-12-21",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2024-12-21",
      "harness": "diff",
      "model": "o1-2024-12-17",
      "modelSlug": "o1-2024-12-17 (high)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 61.7,
      "passRate": 61.7,
      "commitHash": "a755079-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/claude-sonnet-4-20250514 (32k thinking)-2025-05-24",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-05-24",
      "harness": "diff",
      "model": "claude-sonnet-4-20250514 (32k thinking)",
      "modelSlug": "claude-sonnet-4-20250514 (32k thinking)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 61.3,
      "passRate": 61.3,
      "commitHash": "e3cb907",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/o3-mini (high)-2025-01-31",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-01-31",
      "harness": "diff",
      "model": "o3-mini",
      "modelSlug": "o3-mini (high)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 60.4,
      "passRate": 60.4,
      "commitHash": "b0d58d1-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/claude-3-7-sonnet-20250219 (no thinking)-2025-02-24",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-02-24",
      "harness": "diff",
      "model": "claude-3-7-sonnet-20250219 (no thinking)",
      "modelSlug": "claude-3-7-sonnet-20250219 (no thinking)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 60.4,
      "passRate": 60.4,
      "commitHash": "75e9ee6",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/Qwen3 235B A22B diff, no think, Alibaba API-2025-05-09",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-05-09",
      "harness": "diff",
      "model": "Qwen3 235B A22B diff, no think, Alibaba API",
      "modelSlug": "Qwen3 235B A22B diff, no think, Alibaba API",
      "resolved": null,
      "total": 225,
      "resolvedPct": 59.6,
      "passRate": 59.6,
      "commitHash": "91d7fbd-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/Kimi K2-2025-07-17",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-07-17",
      "harness": "diff",
      "model": "Kimi K2",
      "modelSlug": "Kimi K2",
      "resolved": null,
      "total": 225,
      "resolvedPct": 59.1,
      "passRate": 59.1,
      "commitHash": "915ebff-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/DeepSeek R1-2025-01-20",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-01-20",
      "harness": "diff",
      "model": "DeepSeek R1",
      "modelSlug": "DeepSeek R1",
      "resolved": null,
      "total": 225,
      "resolvedPct": 56.9,
      "passRate": 56.9,
      "commitHash": "5650697-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/claude-sonnet-4-20250514 (no thinking)-2025-05-24",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-05-24",
      "harness": "diff",
      "model": "claude-sonnet-4-20250514 (no thinking)",
      "modelSlug": "claude-sonnet-4-20250514 (no thinking)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 56.4,
      "passRate": 56.4,
      "commitHash": "ef3f8bb-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/DeepSeek V3 (0324)-2025-03-24",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-03-24",
      "harness": "diff",
      "model": "DeepSeek V3 (0324)",
      "modelSlug": "DeepSeek V3 (0324)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 55.1,
      "passRate": 55.1,
      "commitHash": "502b863",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/gemini-2.5-flash-preview-05-20 (24k think)-2025-05-25",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-05-25",
      "harness": "diff",
      "model": "gemini-2.5-flash-preview-05-20 (24k think)",
      "modelSlug": "gemini-2.5-flash-preview-05-20 (24k think)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 55.1,
      "passRate": 55.1,
      "commitHash": "a8568c3-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/Quasar Alpha-2025-04-04",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-04-04",
      "harness": "diff",
      "model": "Quasar Alpha",
      "modelSlug": "Quasar Alpha",
      "resolved": null,
      "total": 225,
      "resolvedPct": 54.7,
      "passRate": 54.7,
      "commitHash": "8a34a6c-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/Grok 3 Beta-2025-04-10",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-04-10",
      "harness": "diff",
      "model": "Grok 3 Beta",
      "modelSlug": "Grok 3 Beta",
      "resolved": null,
      "total": 225,
      "resolvedPct": 53.3,
      "passRate": 53.3,
      "commitHash": "2dd40fc-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/Optimus Alpha-2025-04-10",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-04-10",
      "harness": "diff",
      "model": "Optimus Alpha",
      "modelSlug": "Optimus Alpha",
      "resolved": null,
      "total": 225,
      "resolvedPct": 52.9,
      "passRate": 52.9,
      "commitHash": "532bc45-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/gpt-4.1-2025-04-14",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-04-14",
      "harness": "diff",
      "model": "gpt-4.1",
      "modelSlug": "gpt-4.1",
      "resolved": null,
      "total": 225,
      "resolvedPct": 52.4,
      "passRate": 52.4,
      "commitHash": "7a87db5-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/claude-3-5-sonnet-20241022-2025-01-17",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-01-17",
      "harness": "diff",
      "model": "claude-3-5-sonnet-20241022",
      "modelSlug": "claude-3-5-sonnet-20241022",
      "resolved": null,
      "total": 225,
      "resolvedPct": 51.6,
      "passRate": 51.6,
      "commitHash": "6451d59",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/Grok 3 Mini Beta (high)-2025-04-10",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-04-10",
      "harness": "whole",
      "model": "Grok 3 Mini Beta",
      "modelSlug": "Grok 3 Mini Beta (high)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 49.3,
      "passRate": 49.3,
      "commitHash": "8ee33da-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/DeepSeek Chat V3 (prev)-2024-12-25",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2024-12-25",
      "harness": "diff",
      "model": "DeepSeek Chat V3 (prev)",
      "modelSlug": "DeepSeek Chat V3 (prev)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 48.4,
      "passRate": 48.4,
      "commitHash": "0a23c4a-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/gemini-2.5-flash-preview-04-17 (default)-2025-04-20",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-04-20",
      "harness": "diff",
      "model": "gemini-2.5-flash-preview-04-17",
      "modelSlug": "gemini-2.5-flash-preview-04-17 (default)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 47.1,
      "passRate": 47.1,
      "commitHash": "7fcce5d-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/chatgpt-4o-latest (2025-03-29)-2025-03-29",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-03-29",
      "harness": "diff",
      "model": "chatgpt-4o-latest (2025-03-29)",
      "modelSlug": "chatgpt-4o-latest (2025-03-29)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 45.3,
      "passRate": 45.3,
      "commitHash": "0decbad",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/gpt-4.5-preview-2025-02-27",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-02-27",
      "harness": "diff",
      "model": "gpt-4.5-preview",
      "modelSlug": "gpt-4.5-preview",
      "resolved": null,
      "total": 225,
      "resolvedPct": 44.9,
      "passRate": 44.9,
      "commitHash": "b462e55-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/gemini-2.5-flash-preview-05-20 (no think)-2025-05-26",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-05-26",
      "harness": "diff",
      "model": "gemini-2.5-flash-preview-05-20 (no think)",
      "modelSlug": "gemini-2.5-flash-preview-05-20 (no think)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 44,
      "passRate": 44,
      "commitHash": "214b811-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/gpt-oss-120b (high)-2025-08-06",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-08-06",
      "harness": "diff",
      "model": "gpt-oss-120b",
      "modelSlug": "gpt-oss-120b (high)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 41.8,
      "passRate": 41.8,
      "commitHash": "1af0e59",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/Qwen3 32B-2025-05-08",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-05-08",
      "harness": "diff",
      "model": "Qwen3 32B",
      "modelSlug": "Qwen3 32B",
      "resolved": null,
      "total": 225,
      "resolvedPct": 40,
      "passRate": 40,
      "commitHash": "aaacee5-dirty, aeaf259",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/gemini-exp-1206-2024-12-22",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2024-12-22",
      "harness": "whole",
      "model": "gemini-exp-1206",
      "modelSlug": "gemini-exp-1206",
      "resolved": null,
      "total": 225,
      "resolvedPct": 38.2,
      "passRate": 38.2,
      "commitHash": "b1bc2f8",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/Gemini 2.0 Pro exp-02-05-2025-02-25",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-02-25",
      "harness": "whole",
      "model": "Gemini 2.0 Pro exp-02-05",
      "modelSlug": "Gemini 2.0 Pro exp-02-05",
      "resolved": null,
      "total": 225,
      "resolvedPct": 35.6,
      "passRate": 35.6,
      "commitHash": "2fccd47",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/o1-mini-2024-09-12-2024-12-22",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2024-12-22",
      "harness": "whole",
      "model": "o1-mini-2024-09-12",
      "modelSlug": "o1-mini-2024-09-12",
      "resolved": null,
      "total": 225,
      "resolvedPct": 32.9,
      "passRate": 32.9,
      "commitHash": "37df899",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/gpt-4.1-mini-2025-04-14",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-04-14",
      "harness": "diff",
      "model": "gpt-4.1-mini",
      "modelSlug": "gpt-4.1-mini",
      "resolved": null,
      "total": 225,
      "resolvedPct": 32.4,
      "passRate": 32.4,
      "commitHash": "ffb743e-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/claude-3-5-haiku-20241022-2024-12-21",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2024-12-21",
      "harness": "diff",
      "model": "claude-3-5-haiku-20241022",
      "modelSlug": "claude-3-5-haiku-20241022",
      "resolved": null,
      "total": 225,
      "resolvedPct": 28,
      "passRate": 28,
      "commitHash": "a755079-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/chatgpt-4o-latest (2025-02-15)-2025-02-15",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-02-15",
      "harness": "diff",
      "model": "chatgpt-4o-latest (2025-02-15)",
      "modelSlug": "chatgpt-4o-latest (2025-02-15)",
      "resolved": null,
      "total": 225,
      "resolvedPct": 27.1,
      "passRate": 27.1,
      "commitHash": "108ce18-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/QwQ-32B + Qwen 2.5 Coder Instruct-2025-03-07",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-03-07",
      "harness": "architect",
      "model": "QwQ-32B + Qwen 2.5 Coder Instruct",
      "modelSlug": "QwQ-32B + Qwen 2.5 Coder Instruct",
      "resolved": null,
      "total": 225,
      "resolvedPct": 26.2,
      "passRate": 26.2,
      "commitHash": "52162a5",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/gpt-4o-2024-08-06-2024-12-30",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2024-12-30",
      "harness": "diff",
      "model": "gpt-4o-2024-08-06",
      "modelSlug": "gpt-4o-2024-08-06",
      "resolved": null,
      "total": 225,
      "resolvedPct": 23.1,
      "passRate": 23.1,
      "commitHash": "09ee197-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/gemini-2.0-flash-exp-2024-12-22",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2024-12-22",
      "harness": "whole",
      "model": "gemini-2.0-flash-exp",
      "modelSlug": "gemini-2.0-flash-exp",
      "resolved": null,
      "total": 225,
      "resolvedPct": 22.2,
      "passRate": 22.2,
      "commitHash": "b1bc2f8",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/qwen-max-2025-01-25-2025-01-28",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-01-28",
      "harness": "diff",
      "model": "qwen-max-2025-01-25",
      "modelSlug": "qwen-max-2025-01-25",
      "resolved": null,
      "total": 225,
      "resolvedPct": 21.8,
      "passRate": 21.8,
      "commitHash": "ae7d459",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/QwQ-32B-2025-03-06",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-03-06",
      "harness": "diff",
      "model": "QwQ-32B",
      "modelSlug": "QwQ-32B",
      "resolved": null,
      "total": 225,
      "resolvedPct": 20.9,
      "passRate": 20.9,
      "commitHash": "51d118f-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/gpt-4o-2024-11-20-2024-12-30",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2024-12-30",
      "harness": "diff",
      "model": "gpt-4o-2024-11-20",
      "modelSlug": "gpt-4o-2024-11-20",
      "resolved": null,
      "total": 225,
      "resolvedPct": 18.2,
      "passRate": 18.2,
      "commitHash": "09ee197-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/gemini-2.0-flash-thinking-exp-01-21-2025-01-21",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-01-21",
      "harness": "diff",
      "model": "gemini-2.0-flash-thinking-exp-01-21",
      "modelSlug": "gemini-2.0-flash-thinking-exp-01-21",
      "resolved": null,
      "total": 225,
      "resolvedPct": 18.2,
      "passRate": 18.2,
      "commitHash": "843720a",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/DeepSeek Chat V2.5-2024-12-21",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2024-12-21",
      "harness": "diff",
      "model": "DeepSeek Chat V2.5",
      "modelSlug": "DeepSeek Chat V2.5",
      "resolved": null,
      "total": 225,
      "resolvedPct": 17.8,
      "passRate": 17.8,
      "commitHash": "a755079-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/Qwen2.5-Coder-32B-Instruct-2024-12-26",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2024-12-26",
      "harness": "whole",
      "model": "Qwen2.5-Coder-32B-Instruct",
      "modelSlug": "Qwen2.5-Coder-32B-Instruct",
      "resolved": null,
      "total": 225,
      "resolvedPct": 16.4,
      "passRate": 16.4,
      "commitHash": "b51768b0",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/Llama 4 Maverick-2025-04-06",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-04-06",
      "harness": "whole",
      "model": "Llama 4 Maverick",
      "modelSlug": "Llama 4 Maverick",
      "resolved": null,
      "total": 225,
      "resolvedPct": 15.6,
      "passRate": 15.6,
      "commitHash": "9445a31",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/yi-lightning-2024-12-23",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2024-12-23",
      "harness": "whole",
      "model": "yi-lightning",
      "modelSlug": "yi-lightning",
      "resolved": null,
      "total": 225,
      "resolvedPct": 12.9,
      "passRate": 12.9,
      "commitHash": "2b1625e",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/command-a-03-2025-quality-2025-03-14",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-03-14",
      "harness": "whole",
      "model": "command-a-03-2025-quality",
      "modelSlug": "command-a-03-2025-quality",
      "resolved": null,
      "total": 225,
      "resolvedPct": 12,
      "passRate": 12,
      "commitHash": "a1aa63f",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/Codestral 25.01-2025-01-13",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-01-13",
      "harness": "whole",
      "model": "Codestral 25.01",
      "modelSlug": "Codestral 25.01",
      "resolved": null,
      "total": 225,
      "resolvedPct": 11.1,
      "passRate": 11.1,
      "commitHash": "0cba898-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/openhands-lm-32b-v0.1-2025-04-19",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-04-19",
      "harness": "whole",
      "model": "openhands-lm-32b-v0.1",
      "modelSlug": "openhands-lm-32b-v0.1",
      "resolved": null,
      "total": 225,
      "resolvedPct": 10.2,
      "passRate": 10.2,
      "commitHash": "c08336f",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/gpt-4.1-nano-2025-04-14",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-04-14",
      "harness": "whole",
      "model": "gpt-4.1-nano",
      "modelSlug": "gpt-4.1-nano",
      "resolved": null,
      "total": 225,
      "resolvedPct": 8.9,
      "passRate": 8.9,
      "commitHash": "71d1591-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/Qwen2.5-Coder-32B-Instruct-2024-12-22",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2024-12-22",
      "harness": "diff",
      "model": "Qwen2.5-Coder-32B-Instruct",
      "modelSlug": "Qwen2.5-Coder-32B-Instruct",
      "resolved": null,
      "total": 225,
      "resolvedPct": 8,
      "passRate": 8,
      "commitHash": "6d7e8be-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/gemma-3-27b-it-2025-03-15",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2025-03-15",
      "harness": "whole",
      "model": "gemma-3-27b-it",
      "modelSlug": "gemma-3-27b-it",
      "resolved": null,
      "total": 225,
      "resolvedPct": 4.9,
      "passRate": 4.9,
      "commitHash": "fd21f51-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "aider/gpt-4o-mini-2024-07-18-2024-12-21",
      "suite": "Aider polyglot",
      "split": "polyglot",
      "date": "2024-12-21",
      "harness": "whole",
      "model": "gpt-4o-mini-2024-07-18",
      "modelSlug": "gpt-4o-mini-2024-07-18",
      "resolved": null,
      "total": 225,
      "resolvedPct": 3.6,
      "passRate": 3.6,
      "commitHash": "a755079-dirty",
      "sourceUrl": "https://aider.chat/docs/leaderboards/"
    },
    {
      "id": "lite/20250911_isea_claude-3.5-sonnet-20241022",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-09-11",
      "harness": "isea",
      "modelSlug": "claude-3.5-sonnet-20241022",
      "model": "Claude 3.5 Sonnet 20241022",
      "resolved": 154,
      "attempted": 154,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 51.3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sympy/sympy": {
          "resolved": 32,
          "total": 77
        },
        "astropy/astropy": {
          "resolved": 3,
          "total": 6
        },
        "matplotlib/matplotlib": {
          "resolved": 13,
          "total": 23
        },
        "psf/requests": {
          "resolved": 1,
          "total": 6
        },
        "pytest-dev/pytest": {
          "resolved": 7,
          "total": 17
        },
        "scikit-learn/scikit-learn": {
          "resolved": 14,
          "total": 23
        },
        "sphinx-doc/sphinx": {
          "resolved": 7,
          "total": 16
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 6
        },
        "mwaskom/seaborn": {
          "resolved": 3,
          "total": 4
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "django/django": {
          "resolved": 69,
          "total": 114
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250911_isea_claude-3.5-sonnet-20241022"
    },
    {
      "id": "lite/20250906_KGCompass_claude-4-sonnet-20250514",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-09-06",
      "harness": "KGCompass",
      "modelSlug": "claude-4-sonnet-20250514",
      "model": "Claude 4 Sonnet 20250514",
      "resolved": 175,
      "attempted": 179,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 58.3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sympy/sympy": {
          "resolved": 40,
          "total": 77
        },
        "matplotlib/matplotlib": {
          "resolved": 11,
          "total": 23
        },
        "pydata/xarray": {
          "resolved": 3,
          "total": 5
        },
        "pytest-dev/pytest": {
          "resolved": 11,
          "total": 17
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 3
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 6
        },
        "scikit-learn/scikit-learn": {
          "resolved": 15,
          "total": 23
        },
        "sphinx-doc/sphinx": {
          "resolved": 2,
          "total": 16
        },
        "psf/requests": {
          "resolved": 5,
          "total": 6
        },
        "astropy/astropy": {
          "resolved": 3,
          "total": 6
        },
        "django/django": {
          "resolved": 79,
          "total": 114
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250906_KGCompass_claude-4-sonnet-20250514"
    },
    {
      "id": "lite/20250901_entroPO_R2E_QwenCoder30BA3B_tts",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-09-01",
      "harness": "entroPO",
      "modelSlug": "R2E_QwenCoder30BA3B_tts",
      "model": "R2E_QwenCoder30BA3B_tts",
      "resolved": 149,
      "attempted": 149,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 49.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "scikit-learn/scikit-learn": {
          "resolved": 17,
          "total": 23
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 6
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "psf/requests": {
          "resolved": 6,
          "total": 6
        },
        "django/django": {
          "resolved": 63,
          "total": 114
        },
        "sphinx-doc/sphinx": {
          "resolved": 7,
          "total": 16
        },
        "pydata/xarray": {
          "resolved": 2,
          "total": 5
        },
        "matplotlib/matplotlib": {
          "resolved": 9,
          "total": 23
        },
        "sympy/sympy": {
          "resolved": 30,
          "total": 77
        },
        "mwaskom/seaborn": {
          "resolved": 4,
          "total": 4
        },
        "astropy/astropy": {
          "resolved": 3,
          "total": 6
        },
        "pytest-dev/pytest": {
          "resolved": 5,
          "total": 17
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250901_entroPO_R2E_QwenCoder30BA3B_tts"
    },
    {
      "id": "lite/20250901_entroPO_R2E_QwenCoder30BA3B",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-09-01",
      "harness": "entroPO",
      "modelSlug": "R2E_QwenCoder30BA3B",
      "model": "R2E_QwenCoder30BA3B",
      "resolved": 135,
      "attempted": 137,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 45,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pydata/xarray": {
          "resolved": 2,
          "total": 5
        },
        "pytest-dev/pytest": {
          "resolved": 5,
          "total": 17
        },
        "sympy/sympy": {
          "resolved": 31,
          "total": 77
        },
        "matplotlib/matplotlib": {
          "resolved": 9,
          "total": 23
        },
        "psf/requests": {
          "resolved": 6,
          "total": 6
        },
        "django/django": {
          "resolved": 55,
          "total": 114
        },
        "sphinx-doc/sphinx": {
          "resolved": 5,
          "total": 16
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 6
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "scikit-learn/scikit-learn": {
          "resolved": 14,
          "total": 23
        },
        "astropy/astropy": {
          "resolved": 3,
          "total": 6
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250901_entroPO_R2E_QwenCoder30BA3B"
    },
    {
      "id": "lite/20250627_agentless_MCTS-Refine-7B",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-06-27",
      "harness": "agentless",
      "modelSlug": "MCTS-Refine-7B",
      "model": "MCTS Refine 7B",
      "resolved": 49,
      "attempted": 87,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 16.3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "django/django": {
          "resolved": 20,
          "total": 114
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 6
        },
        "psf/requests": {
          "resolved": 1,
          "total": 6
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 4
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "matplotlib/matplotlib": {
          "resolved": 4,
          "total": 23
        },
        "scikit-learn/scikit-learn": {
          "resolved": 4,
          "total": 23
        },
        "sphinx-doc/sphinx": {
          "resolved": 0,
          "total": 16
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "sympy/sympy": {
          "resolved": 9,
          "total": 77
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "pytest-dev/pytest": {
          "resolved": 5,
          "total": 17
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250627_agentless_MCTS-Refine-7B"
    },
    {
      "id": "lite/20250625_ExpeRepair-v1_claude-4-sonnet-20250514",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-06-25",
      "harness": "ExpeRepair-v1",
      "modelSlug": "claude-4-sonnet-20250514",
      "model": "Claude 4 Sonnet 20250514",
      "resolved": 181,
      "attempted": 181,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 60.3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "astropy/astropy": {
          "resolved": 3,
          "total": 6
        },
        "django/django": {
          "resolved": 79,
          "total": 114
        },
        "matplotlib/matplotlib": {
          "resolved": 12,
          "total": 23
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "psf/requests": {
          "resolved": 3,
          "total": 6
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 6
        },
        "sympy/sympy": {
          "resolved": 41,
          "total": 77
        },
        "sphinx-doc/sphinx": {
          "resolved": 9,
          "total": 16
        },
        "pytest-dev/pytest": {
          "resolved": 10,
          "total": 17
        },
        "pydata/xarray": {
          "resolved": 2,
          "total": 5
        },
        "scikit-learn/scikit-learn": {
          "resolved": 16,
          "total": 23
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250625_ExpeRepair-v1_claude-4-sonnet-20250514"
    },
    {
      "id": "lite/20250625_SemAgent_Multi-v1_Claude3.7Sonnet_Gemini2.5Pro",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-06-25",
      "harness": "SemAgent",
      "modelSlug": "Multi-v1_Claude3.7Sonnet_Gemini2.5Pro",
      "model": "Multi V1_Claude3.7Sonnet_Gemini2.5Pro",
      "resolved": 155,
      "attempted": 155,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 51.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pytest-dev/pytest": {
          "resolved": 8,
          "total": 17
        },
        "astropy/astropy": {
          "resolved": 3,
          "total": 6
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 3
        },
        "sympy/sympy": {
          "resolved": 33,
          "total": 77
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 4
        },
        "scikit-learn/scikit-learn": {
          "resolved": 14,
          "total": 23
        },
        "psf/requests": {
          "resolved": 5,
          "total": 6
        },
        "matplotlib/matplotlib": {
          "resolved": 13,
          "total": 23
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "django/django": {
          "resolved": 65,
          "total": 114
        },
        "sphinx-doc/sphinx": {
          "resolved": 8,
          "total": 16
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 6
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250625_SemAgent_Multi-v1_Claude3.7Sonnet_Gemini2.5Pro"
    },
    {
      "id": "lite/20250619_KGCompass_claude-3.5-sonnet-20241022",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-06-19",
      "harness": "KGCompass",
      "modelSlug": "claude-3.5-sonnet-20241022",
      "model": "Claude 3.5 Sonnet 20241022",
      "resolved": 138,
      "attempted": 147,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 46,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sympy/sympy": {
          "resolved": 29,
          "total": 77
        },
        "scikit-learn/scikit-learn": {
          "resolved": 12,
          "total": 23
        },
        "django/django": {
          "resolved": 64,
          "total": 114
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "psf/requests": {
          "resolved": 5,
          "total": 6
        },
        "sphinx-doc/sphinx": {
          "resolved": 3,
          "total": 16
        },
        "matplotlib/matplotlib": {
          "resolved": 8,
          "total": 23
        },
        "astropy/astropy": {
          "resolved": 3,
          "total": 6
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 3
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 6
        },
        "pytest-dev/pytest": {
          "resolved": 9,
          "total": 17
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250619_KGCompass_claude-3.5-sonnet-20241022"
    },
    {
      "id": "lite/20250609_KGCompass_deepseek-v3",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-06-09",
      "harness": "KGCompass",
      "modelSlug": "deepseek-v3",
      "model": "DeepSeek V3",
      "resolved": 110,
      "attempted": 116,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 36.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "django/django": {
          "resolved": 52,
          "total": 114
        },
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 6
        },
        "scikit-learn/scikit-learn": {
          "resolved": 10,
          "total": 23
        },
        "sympy/sympy": {
          "resolved": 23,
          "total": 77
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 3
        },
        "astropy/astropy": {
          "resolved": 3,
          "total": 6
        },
        "sphinx-doc/sphinx": {
          "resolved": 1,
          "total": 16
        },
        "psf/requests": {
          "resolved": 3,
          "total": 6
        },
        "matplotlib/matplotlib": {
          "resolved": 5,
          "total": 23
        },
        "pytest-dev/pytest": {
          "resolved": 8,
          "total": 17
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250609_KGCompass_deepseek-v3"
    },
    {
      "id": "lite/20250526_sweagent_claude-4-sonnet-20250514",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-05-26",
      "harness": "sweagent",
      "modelSlug": "claude-4-sonnet-20250514",
      "model": "Claude 4 Sonnet 20250514",
      "resolved": 170,
      "attempted": 171,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 56.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "scikit-learn/scikit-learn": {
          "resolved": 15,
          "total": 23
        },
        "pydata/xarray": {
          "resolved": 2,
          "total": 5
        },
        "psf/requests": {
          "resolved": 3,
          "total": 6
        },
        "sympy/sympy": {
          "resolved": 40,
          "total": 77
        },
        "matplotlib/matplotlib": {
          "resolved": 9,
          "total": 23
        },
        "astropy/astropy": {
          "resolved": 3,
          "total": 6
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 6
        },
        "django/django": {
          "resolved": 75,
          "total": 114
        },
        "pytest-dev/pytest": {
          "resolved": 10,
          "total": 17
        },
        "sphinx-doc/sphinx": {
          "resolved": 8,
          "total": 16
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250526_sweagent_claude-4-sonnet-20250514"
    },
    {
      "id": "lite/20250509_Lingxi_claude-3-5-sonnet-20241022",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-05-09",
      "harness": "Lingxi",
      "modelSlug": "claude-3-5-sonnet-20241022",
      "model": "Claude 3.5 Sonnet 20241022",
      "resolved": 128,
      "attempted": 129,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 42.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "matplotlib/matplotlib": {
          "resolved": 10,
          "total": 23
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "psf/requests": {
          "resolved": 1,
          "total": 6
        },
        "sphinx-doc/sphinx": {
          "resolved": 0,
          "total": 16
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "sympy/sympy": {
          "resolved": 26,
          "total": 77
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 6
        },
        "scikit-learn/scikit-learn": {
          "resolved": 12,
          "total": 23
        },
        "django/django": {
          "resolved": 63,
          "total": 114
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "pytest-dev/pytest": {
          "resolved": 7,
          "total": 17
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250509_Lingxi_claude-3-5-sonnet-20241022"
    },
    {
      "id": "lite/20250425_Refact_Agent",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-04-25",
      "harness": "Refact",
      "modelSlug": "Agent",
      "model": "Agent",
      "resolved": 180,
      "attempted": 181,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 60,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "matplotlib/matplotlib": {
          "resolved": 11,
          "total": 23
        },
        "pytest-dev/pytest": {
          "resolved": 10,
          "total": 17
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "pydata/xarray": {
          "resolved": 2,
          "total": 5
        },
        "scikit-learn/scikit-learn": {
          "resolved": 17,
          "total": 23
        },
        "astropy/astropy": {
          "resolved": 3,
          "total": 6
        },
        "psf/requests": {
          "resolved": 5,
          "total": 6
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 6
        },
        "sphinx-doc/sphinx": {
          "resolved": 6,
          "total": 16
        },
        "django/django": {
          "resolved": 78,
          "total": 114
        },
        "sympy/sympy": {
          "resolved": 43,
          "total": 77
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250425_Refact_Agent"
    },
    {
      "id": "lite/20250306_SWE-Fixer_Qwen2.5-7b-retriever_Qwen2.5-72b-editor",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-03-06",
      "harness": "SWE-Fixer",
      "modelSlug": "Qwen2.5-7b-retriever_Qwen2.5-72b-editor",
      "model": "Qwen2.5 7b Retriever_Qwen2.5 72b Editor",
      "resolved": 74,
      "attempted": 78,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 24.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "psf/requests": {
          "resolved": 3,
          "total": 6
        },
        "scikit-learn/scikit-learn": {
          "resolved": 6,
          "total": 23
        },
        "matplotlib/matplotlib": {
          "resolved": 5,
          "total": 23
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "sphinx-doc/sphinx": {
          "resolved": 4,
          "total": 16
        },
        "django/django": {
          "resolved": 37,
          "total": 114
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 6
        },
        "sympy/sympy": {
          "resolved": 12,
          "total": 77
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 4
        },
        "pytest-dev/pytest": {
          "resolved": 2,
          "total": 17
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250306_SWE-Fixer_Qwen2.5-7b-retriever_Qwen2.5-72b-editor"
    },
    {
      "id": "lite/20250226_sweagent_claude-3-7-sonnet-20250219",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-02-26",
      "harness": "sweagent",
      "modelSlug": "claude-3-7-sonnet-20250219",
      "model": "Claude 3.7 Sonnet 20250219",
      "resolved": 144,
      "attempted": 146,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 48,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pytest-dev/pytest": {
          "resolved": 7,
          "total": 17
        },
        "sphinx-doc/sphinx": {
          "resolved": 6,
          "total": 16
        },
        "django/django": {
          "resolved": 62,
          "total": 114
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "matplotlib/matplotlib": {
          "resolved": 13,
          "total": 23
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "psf/requests": {
          "resolved": 3,
          "total": 6
        },
        "sympy/sympy": {
          "resolved": 30,
          "total": 77
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 6
        },
        "scikit-learn/scikit-learn": {
          "resolved": 15,
          "total": 23
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250226_sweagent_claude-3-7-sonnet-20250219"
    },
    {
      "id": "lite/20250214_agentless_lite_o3_mini",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-02-14",
      "harness": "agentless",
      "modelSlug": "lite_o3_mini",
      "model": "Lite_o3_mini",
      "resolved": 97,
      "attempted": 103,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 32.3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "django/django": {
          "resolved": 49,
          "total": 114
        },
        "psf/requests": {
          "resolved": 3,
          "total": 6
        },
        "sphinx-doc/sphinx": {
          "resolved": 6,
          "total": 16
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 6
        },
        "sympy/sympy": {
          "resolved": 20,
          "total": 77
        },
        "pytest-dev/pytest": {
          "resolved": 4,
          "total": 17
        },
        "scikit-learn/scikit-learn": {
          "resolved": 5,
          "total": 23
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 4
        },
        "matplotlib/matplotlib": {
          "resolved": 4,
          "total": 23
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250214_agentless_lite_o3_mini"
    },
    {
      "id": "lite/20250207_aegis_o3mini",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-02-07",
      "harness": "aegis",
      "modelSlug": "o3mini",
      "model": "O3mini",
      "resolved": 91,
      "attempted": 130,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 30.3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "sympy/sympy": {
          "resolved": 16,
          "total": 77
        },
        "sphinx-doc/sphinx": {
          "resolved": 3,
          "total": 16
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "pytest-dev/pytest": {
          "resolved": 3,
          "total": 17
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 6
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "django/django": {
          "resolved": 42,
          "total": 114
        },
        "scikit-learn/scikit-learn": {
          "resolved": 10,
          "total": 23
        },
        "matplotlib/matplotlib": {
          "resolved": 8,
          "total": 23
        },
        "psf/requests": {
          "resolved": 2,
          "total": 6
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250207_aegis_o3mini"
    },
    {
      "id": "lite/20250205_dars_agent_claude_3.5_sonnet_deepseek_r1",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-02-05",
      "harness": "dars",
      "modelSlug": "agent_claude_3.5_sonnet_deepseek_r1",
      "model": "Agent_claude_3.5_sonnet_deepseek_r1",
      "resolved": 141,
      "attempted": 141,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 47,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "mwaskom/seaborn": {
          "resolved": 3,
          "total": 4
        },
        "astropy/astropy": {
          "resolved": 3,
          "total": 6
        },
        "sympy/sympy": {
          "resolved": 27,
          "total": 77
        },
        "pytest-dev/pytest": {
          "resolved": 6,
          "total": 17
        },
        "matplotlib/matplotlib": {
          "resolved": 11,
          "total": 23
        },
        "sphinx-doc/sphinx": {
          "resolved": 0,
          "total": 16
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 6
        },
        "django/django": {
          "resolved": 65,
          "total": 114
        },
        "scikit-learn/scikit-learn": {
          "resolved": 16,
          "total": 23
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "pydata/xarray": {
          "resolved": 2,
          "total": 5
        },
        "psf/requests": {
          "resolved": 4,
          "total": 6
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250205_dars_agent_claude_3.5_sonnet_deepseek_r1"
    },
    {
      "id": "lite/20250114_moatless_claude-3.5-sonnet-20241022",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-01-14",
      "harness": "moatless",
      "modelSlug": "claude-3.5-sonnet-20241022",
      "model": "Claude 3.5 Sonnet 20241022",
      "resolved": 117,
      "attempted": 128,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 39,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 4
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "sphinx-doc/sphinx": {
          "resolved": 5,
          "total": 16
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "django/django": {
          "resolved": 51,
          "total": 114
        },
        "pytest-dev/pytest": {
          "resolved": 5,
          "total": 17
        },
        "psf/requests": {
          "resolved": 4,
          "total": 6
        },
        "scikit-learn/scikit-learn": {
          "resolved": 13,
          "total": 23
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 6
        },
        "sympy/sympy": {
          "resolved": 22,
          "total": 77
        },
        "matplotlib/matplotlib": {
          "resolved": 10,
          "total": 23
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250114_moatless_claude-3.5-sonnet-20241022"
    },
    {
      "id": "lite/20250113_OpenCSG-Starship-Agentic-Coder_gpt4o",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-01-13",
      "harness": "OpenCSG-Starship-Agentic-Coder",
      "modelSlug": "gpt4o",
      "model": "Gpt4o",
      "resolved": 119,
      "attempted": 133,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 39.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "psf/requests": {
          "resolved": 0,
          "total": 6
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 6
        },
        "sphinx-doc/sphinx": {
          "resolved": 4,
          "total": 16
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "astropy/astropy": {
          "resolved": 3,
          "total": 6
        },
        "pytest-dev/pytest": {
          "resolved": 5,
          "total": 17
        },
        "sympy/sympy": {
          "resolved": 26,
          "total": 77
        },
        "scikit-learn/scikit-learn": {
          "resolved": 12,
          "total": 23
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "matplotlib/matplotlib": {
          "resolved": 8,
          "total": 23
        },
        "django/django": {
          "resolved": 56,
          "total": 114
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250113_OpenCSG-Starship-Agentic-Coder_gpt4o"
    },
    {
      "id": "lite/20250111_moatless_deepseek_v3",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-01-11",
      "harness": "moatless",
      "modelSlug": "deepseek_v3",
      "model": "Deepseek_v3",
      "resolved": 92,
      "attempted": 105,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 30.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sympy/sympy": {
          "resolved": 19,
          "total": 77
        },
        "psf/requests": {
          "resolved": 2,
          "total": 6
        },
        "sphinx-doc/sphinx": {
          "resolved": 2,
          "total": 16
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "astropy/astropy": {
          "resolved": 1,
          "total": 6
        },
        "matplotlib/matplotlib": {
          "resolved": 7,
          "total": 23
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 6
        },
        "scikit-learn/scikit-learn": {
          "resolved": 13,
          "total": 23
        },
        "pydata/xarray": {
          "resolved": 0,
          "total": 5
        },
        "django/django": {
          "resolved": 40,
          "total": 114
        },
        "pytest-dev/pytest": {
          "resolved": 4,
          "total": 17
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250111_moatless_deepseek_v3"
    },
    {
      "id": "lite/20250104_patched_codes_claude-3.5-sonnet-20241022",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2025-01-04",
      "harness": "patched",
      "modelSlug": "codes_claude-3.5-sonnet-20241022",
      "model": "Codes_claude 3.5 Sonnet 20241022",
      "resolved": 111,
      "attempted": 224,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 37,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "django/django": {
          "resolved": 58,
          "total": 114
        },
        "scikit-learn/scikit-learn": {
          "resolved": 11,
          "total": 23
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 6
        },
        "sphinx-doc/sphinx": {
          "resolved": 3,
          "total": 16
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "sympy/sympy": {
          "resolved": 21,
          "total": 77
        },
        "pytest-dev/pytest": {
          "resolved": 6,
          "total": 17
        },
        "matplotlib/matplotlib": {
          "resolved": 0,
          "total": 23
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "psf/requests": {
          "resolved": 4,
          "total": 6
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20250104_patched_codes_claude-3.5-sonnet-20241022"
    },
    {
      "id": "lite/20241220_blackboxai_agent_v1",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-12-20",
      "harness": "blackboxai",
      "modelSlug": "agent_v1",
      "model": "Agent_v1",
      "resolved": 147,
      "attempted": 159,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 49,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "astropy/astropy": {
          "resolved": 3,
          "total": 6
        },
        "matplotlib/matplotlib": {
          "resolved": 9,
          "total": 23
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "pydata/xarray": {
          "resolved": 2,
          "total": 5
        },
        "django/django": {
          "resolved": 63,
          "total": 114
        },
        "sympy/sympy": {
          "resolved": 33,
          "total": 77
        },
        "sphinx-doc/sphinx": {
          "resolved": 7,
          "total": 16
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 4
        },
        "psf/requests": {
          "resolved": 5,
          "total": 6
        },
        "scikit-learn/scikit-learn": {
          "resolved": 13,
          "total": 23
        },
        "pytest-dev/pytest": {
          "resolved": 7,
          "total": 17
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 6
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20241220_blackboxai_agent_v1"
    },
    {
      "id": "lite/20241220_PatchKitty-0.9_claude-3.5-sonnet-20241022",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-12-20",
      "harness": "PatchKitty-0.9",
      "modelSlug": "claude-3.5-sonnet-20241022",
      "model": "Claude 3.5 Sonnet 20241022",
      "resolved": 124,
      "attempted": 129,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 41.3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "psf/requests": {
          "resolved": 4,
          "total": 6
        },
        "pytest-dev/pytest": {
          "resolved": 5,
          "total": 17
        },
        "sphinx-doc/sphinx": {
          "resolved": 4,
          "total": 16
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 6
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "matplotlib/matplotlib": {
          "resolved": 9,
          "total": 23
        },
        "sympy/sympy": {
          "resolved": 25,
          "total": 77
        },
        "scikit-learn/scikit-learn": {
          "resolved": 11,
          "total": 23
        },
        "django/django": {
          "resolved": 58,
          "total": 114
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20241220_PatchKitty-0.9_claude-3.5-sonnet-20241022"
    },
    {
      "id": "lite/20241207_kodu_sonnet_v1",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-12-07",
      "harness": "kodu",
      "modelSlug": "sonnet_v1",
      "model": "Sonnet_v1",
      "resolved": 134,
      "attempted": 208,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 44.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 3
        },
        "pydata/xarray": {
          "resolved": 2,
          "total": 5
        },
        "django/django": {
          "resolved": 65,
          "total": 114
        },
        "psf/requests": {
          "resolved": 5,
          "total": 6
        },
        "pytest-dev/pytest": {
          "resolved": 7,
          "total": 17
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 6
        },
        "matplotlib/matplotlib": {
          "resolved": 9,
          "total": 23
        },
        "astropy/astropy": {
          "resolved": 0,
          "total": 6
        },
        "scikit-learn/scikit-learn": {
          "resolved": 8,
          "total": 23
        },
        "sphinx-doc/sphinx": {
          "resolved": 2,
          "total": 16
        },
        "sympy/sympy": {
          "resolved": 29,
          "total": 77
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20241207_kodu_sonnet_v1"
    },
    {
      "id": "lite/20241202_agentless-1.5_claude-3.5-sonnet-20241022",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-12-02",
      "harness": "agentless-1.5",
      "modelSlug": "claude-3.5-sonnet-20241022",
      "model": "Claude 3.5 Sonnet 20241022",
      "resolved": 122,
      "attempted": 124,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 40.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sphinx-doc/sphinx": {
          "resolved": 3,
          "total": 16
        },
        "django/django": {
          "resolved": 53,
          "total": 114
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 6
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "psf/requests": {
          "resolved": 4,
          "total": 6
        },
        "scikit-learn/scikit-learn": {
          "resolved": 13,
          "total": 23
        },
        "sympy/sympy": {
          "resolved": 26,
          "total": 77
        },
        "astropy/astropy": {
          "resolved": 1,
          "total": 6
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "matplotlib/matplotlib": {
          "resolved": 10,
          "total": 23
        },
        "pytest-dev/pytest": {
          "resolved": 6,
          "total": 17
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20241202_agentless-1.5_claude-3.5-sonnet-20241022"
    },
    {
      "id": "lite/20241128_SWE-Fixer_Qwen2.5-7b-retriever_Qwen2.5-72b-editor_20241128",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-11-28",
      "harness": "SWE-Fixer",
      "modelSlug": "Qwen2.5-7b-retriever_Qwen2.5-72b-editor_20241128",
      "model": "Qwen2.5 7b Retriever_Qwen2.5 72b Editor_20241128",
      "resolved": 70,
      "attempted": 70,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 23.3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "django/django": {
          "resolved": 35,
          "total": 114
        },
        "pytest-dev/pytest": {
          "resolved": 1,
          "total": 17
        },
        "scikit-learn/scikit-learn": {
          "resolved": 6,
          "total": 23
        },
        "matplotlib/matplotlib": {
          "resolved": 5,
          "total": 23
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "sympy/sympy": {
          "resolved": 11,
          "total": 77
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 4
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "psf/requests": {
          "resolved": 3,
          "total": 6
        },
        "sphinx-doc/sphinx": {
          "resolved": 4,
          "total": 16
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 6
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20241128_SWE-Fixer_Qwen2.5-7b-retriever_Qwen2.5-72b-editor_20241128"
    },
    {
      "id": "lite/20241127_globant_codefixer_agent",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-11-27",
      "harness": "globant",
      "modelSlug": "codefixer_agent",
      "model": "Codefixer_agent",
      "resolved": 145,
      "attempted": 160,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 48.3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sphinx-doc/sphinx": {
          "resolved": 4,
          "total": 16
        },
        "astropy/astropy": {
          "resolved": 3,
          "total": 6
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 6
        },
        "psf/requests": {
          "resolved": 4,
          "total": 6
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "pytest-dev/pytest": {
          "resolved": 9,
          "total": 17
        },
        "sympy/sympy": {
          "resolved": 31,
          "total": 77
        },
        "scikit-learn/scikit-learn": {
          "resolved": 15,
          "total": 23
        },
        "django/django": {
          "resolved": 65,
          "total": 114
        },
        "matplotlib/matplotlib": {
          "resolved": 7,
          "total": 23
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20241127_globant_codefixer_agent"
    },
    {
      "id": "lite/20241117_moatless_claude-3.5-sonnet-20241022",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-11-17",
      "harness": "moatless",
      "modelSlug": "claude-3.5-sonnet-20241022",
      "model": "Claude 3.5 Sonnet 20241022",
      "resolved": 115,
      "attempted": 126,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 38.3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pytest-dev/pytest": {
          "resolved": 5,
          "total": 17
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "psf/requests": {
          "resolved": 4,
          "total": 6
        },
        "astropy/astropy": {
          "resolved": 3,
          "total": 6
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "scikit-learn/scikit-learn": {
          "resolved": 13,
          "total": 23
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 6
        },
        "sphinx-doc/sphinx": {
          "resolved": 4,
          "total": 16
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "matplotlib/matplotlib": {
          "resolved": 9,
          "total": 23
        },
        "django/django": {
          "resolved": 49,
          "total": 114
        },
        "sympy/sympy": {
          "resolved": 23,
          "total": 77
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20241117_moatless_claude-3.5-sonnet-20241022"
    },
    {
      "id": "lite/20241117_reproducedRG_gpt4o",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-11-17",
      "harness": "reproducedRG",
      "modelSlug": "gpt4o",
      "model": "Gpt4o",
      "resolved": 84,
      "attempted": 91,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 28,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sympy/sympy": {
          "resolved": 17,
          "total": 77
        },
        "pytest-dev/pytest": {
          "resolved": 4,
          "total": 17
        },
        "matplotlib/matplotlib": {
          "resolved": 7,
          "total": 23
        },
        "sphinx-doc/sphinx": {
          "resolved": 3,
          "total": 16
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 4
        },
        "psf/requests": {
          "resolved": 0,
          "total": 6
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "scikit-learn/scikit-learn": {
          "resolved": 10,
          "total": 23
        },
        "django/django": {
          "resolved": 39,
          "total": 114
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "pylint-dev/pylint": {
          "resolved": 0,
          "total": 6
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20241117_reproducedRG_gpt4o"
    },
    {
      "id": "lite/20241111_codeshelltester_gpt4o",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-11-11",
      "harness": "codeshelltester",
      "modelSlug": "gpt4o",
      "model": "Gpt4o",
      "resolved": 94,
      "attempted": 106,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 31.3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pytest-dev/pytest": {
          "resolved": 6,
          "total": 17
        },
        "matplotlib/matplotlib": {
          "resolved": 10,
          "total": 23
        },
        "sphinx-doc/sphinx": {
          "resolved": 3,
          "total": 16
        },
        "pydata/xarray": {
          "resolved": 0,
          "total": 5
        },
        "scikit-learn/scikit-learn": {
          "resolved": 10,
          "total": 23
        },
        "sympy/sympy": {
          "resolved": 20,
          "total": 77
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "psf/requests": {
          "resolved": 2,
          "total": 6
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "astropy/astropy": {
          "resolved": 1,
          "total": 6
        },
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 6
        },
        "django/django": {
          "resolved": 39,
          "total": 114
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20241111_codeshelltester_gpt4o"
    },
    {
      "id": "lite/20241030_composio_swekit",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-10-30",
      "harness": "composio",
      "modelSlug": "swekit",
      "model": "Swekit",
      "resolved": 123,
      "attempted": 124,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 41,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "psf/requests": {
          "resolved": 1,
          "total": 6
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 4
        },
        "scikit-learn/scikit-learn": {
          "resolved": 14,
          "total": 23
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "sympy/sympy": {
          "resolved": 26,
          "total": 77
        },
        "matplotlib/matplotlib": {
          "resolved": 10,
          "total": 23
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "django/django": {
          "resolved": 54,
          "total": 114
        },
        "pytest-dev/pytest": {
          "resolved": 7,
          "total": 17
        },
        "sphinx-doc/sphinx": {
          "resolved": 4,
          "total": 16
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 6
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20241030_composio_swekit"
    },
    {
      "id": "lite/20241028_agentless-1.5_gpt4o",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-10-28",
      "harness": "agentless-1.5",
      "modelSlug": "gpt4o",
      "model": "Gpt4o",
      "resolved": 96,
      "attempted": 100,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 32,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pytest-dev/pytest": {
          "resolved": 6,
          "total": 17
        },
        "scikit-learn/scikit-learn": {
          "resolved": 11,
          "total": 23
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "sympy/sympy": {
          "resolved": 18,
          "total": 77
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "psf/requests": {
          "resolved": 3,
          "total": 6
        },
        "pydata/xarray": {
          "resolved": 0,
          "total": 5
        },
        "django/django": {
          "resolved": 42,
          "total": 114
        },
        "sphinx-doc/sphinx": {
          "resolved": 3,
          "total": 16
        },
        "matplotlib/matplotlib": {
          "resolved": 8,
          "total": 23
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 6
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 4
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20241028_agentless-1.5_gpt4o"
    },
    {
      "id": "lite/20240925_hyperagent_lite1",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-09-25",
      "harness": "hyperagent",
      "modelSlug": "lite1",
      "model": "Lite1",
      "resolved": 76,
      "attempted": 119,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 25.3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sphinx-doc/sphinx": {
          "resolved": 0,
          "total": 16
        },
        "psf/requests": {
          "resolved": 1,
          "total": 6
        },
        "django/django": {
          "resolved": 44,
          "total": 114
        },
        "matplotlib/matplotlib": {
          "resolved": 4,
          "total": 23
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "pytest-dev/pytest": {
          "resolved": 4,
          "total": 17
        },
        "sympy/sympy": {
          "resolved": 10,
          "total": 77
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "astropy/astropy": {
          "resolved": 1,
          "total": 6
        },
        "pylint-dev/pylint": {
          "resolved": 0,
          "total": 6
        },
        "scikit-learn/scikit-learn": {
          "resolved": 9,
          "total": 23
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240925_hyperagent_lite1"
    },
    {
      "id": "lite/20240908_infant_gpt4o",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-09-08",
      "harness": "infant",
      "modelSlug": "gpt4o",
      "model": "Gpt4o",
      "resolved": 90,
      "attempted": 117,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 30,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pydata/xarray": {
          "resolved": 2,
          "total": 5
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "pytest-dev/pytest": {
          "resolved": 4,
          "total": 17
        },
        "sphinx-doc/sphinx": {
          "resolved": 3,
          "total": 16
        },
        "sympy/sympy": {
          "resolved": 21,
          "total": 77
        },
        "psf/requests": {
          "resolved": 5,
          "total": 6
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 4
        },
        "scikit-learn/scikit-learn": {
          "resolved": 7,
          "total": 23
        },
        "django/django": {
          "resolved": 40,
          "total": 114
        },
        "matplotlib/matplotlib": {
          "resolved": 5,
          "total": 23
        },
        "pylint-dev/pylint": {
          "resolved": 0,
          "total": 6
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240908_infant_gpt4o"
    },
    {
      "id": "lite/20240828_autose_mixed",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-08-28",
      "harness": "autose",
      "modelSlug": "mixed",
      "model": "Mixed",
      "resolved": 65,
      "attempted": 96,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 21.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pytest-dev/pytest": {
          "resolved": 5,
          "total": 17
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "sphinx-doc/sphinx": {
          "resolved": 0,
          "total": 16
        },
        "pylint-dev/pylint": {
          "resolved": 0,
          "total": 6
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 4
        },
        "psf/requests": {
          "resolved": 3,
          "total": 6
        },
        "scikit-learn/scikit-learn": {
          "resolved": 7,
          "total": 23
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "django/django": {
          "resolved": 35,
          "total": 114
        },
        "astropy/astropy": {
          "resolved": 1,
          "total": 6
        },
        "sympy/sympy": {
          "resolved": 10,
          "total": 77
        },
        "matplotlib/matplotlib": {
          "resolved": 2,
          "total": 23
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240828_autose_mixed"
    },
    {
      "id": "lite/20240808_RepoGraph_gpt4o",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-08-08",
      "harness": "RepoGraph",
      "modelSlug": "gpt4o",
      "model": "Gpt4o",
      "resolved": 89,
      "attempted": 96,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 29.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sympy/sympy": {
          "resolved": 16,
          "total": 77
        },
        "django/django": {
          "resolved": 39,
          "total": 114
        },
        "pytest-dev/pytest": {
          "resolved": 7,
          "total": 17
        },
        "sphinx-doc/sphinx": {
          "resolved": 3,
          "total": 16
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "matplotlib/matplotlib": {
          "resolved": 6,
          "total": 23
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "pydata/xarray": {
          "resolved": 0,
          "total": 5
        },
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 6
        },
        "psf/requests": {
          "resolved": 4,
          "total": 6
        },
        "scikit-learn/scikit-learn": {
          "resolved": 10,
          "total": 23
        },
        "astropy/astropy": {
          "resolved": 1,
          "total": 6
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240808_RepoGraph_gpt4o"
    },
    {
      "id": "lite/20240728_sweagent_gpt4o",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-07-28",
      "harness": "sweagent",
      "modelSlug": "gpt4o",
      "model": "Gpt4o",
      "resolved": 55,
      "attempted": 94,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 18.3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sympy/sympy": {
          "resolved": 12,
          "total": 77
        },
        "django/django": {
          "resolved": 26,
          "total": 114
        },
        "psf/requests": {
          "resolved": 2,
          "total": 6
        },
        "pylint-dev/pylint": {
          "resolved": 0,
          "total": 6
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "pytest-dev/pytest": {
          "resolved": 5,
          "total": 17
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "scikit-learn/scikit-learn": {
          "resolved": 6,
          "total": 23
        },
        "sphinx-doc/sphinx": {
          "resolved": 0,
          "total": 16
        },
        "matplotlib/matplotlib": {
          "resolved": 0,
          "total": 23
        },
        "astropy/astropy": {
          "resolved": 1,
          "total": 6
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240728_sweagent_gpt4o"
    },
    {
      "id": "lite/20240725_opendevin_codeact_v1.8_claude35sonnet",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-07-25",
      "harness": "opendevin",
      "modelSlug": "codeact_v1.8_claude35sonnet",
      "model": "Codeact_v1.8_claude35sonnet",
      "resolved": 80,
      "attempted": 113,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 26.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "django/django": {
          "resolved": 46,
          "total": 114
        },
        "psf/requests": {
          "resolved": 3,
          "total": 6
        },
        "scikit-learn/scikit-learn": {
          "resolved": 5,
          "total": 23
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "pylint-dev/pylint": {
          "resolved": 0,
          "total": 6
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "sphinx-doc/sphinx": {
          "resolved": 3,
          "total": 16
        },
        "matplotlib/matplotlib": {
          "resolved": 3,
          "total": 23
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "pydata/xarray": {
          "resolved": 0,
          "total": 5
        },
        "sympy/sympy": {
          "resolved": 13,
          "total": 77
        },
        "pytest-dev/pytest": {
          "resolved": 3,
          "total": 17
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240725_opendevin_codeact_v1.8_claude35sonnet"
    },
    {
      "id": "lite/20240706_sima_gpt4o",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-07-06",
      "harness": "sima",
      "modelSlug": "gpt4o",
      "model": "Gpt4o",
      "resolved": 83,
      "attempted": 300,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 27.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "psf/requests": {
          "resolved": 4,
          "total": 6
        },
        "sympy/sympy": {
          "resolved": 18,
          "total": 77
        },
        "scikit-learn/scikit-learn": {
          "resolved": 9,
          "total": 23
        },
        "django/django": {
          "resolved": 37,
          "total": 114
        },
        "pytest-dev/pytest": {
          "resolved": 4,
          "total": 17
        },
        "sphinx-doc/sphinx": {
          "resolved": 3,
          "total": 16
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 6
        },
        "matplotlib/matplotlib": {
          "resolved": 4,
          "total": 23
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 4
        },
        "astropy/astropy": {
          "resolved": 1,
          "total": 6
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240706_sima_gpt4o"
    },
    {
      "id": "lite/20240702_codestory_aide_mixed",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-07-02",
      "harness": "codestory",
      "modelSlug": "aide_mixed",
      "model": "Aide_mixed",
      "resolved": 129,
      "attempted": 273,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 43,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "django/django": {
          "resolved": 60,
          "total": 114
        },
        "astropy/astropy": {
          "resolved": 5,
          "total": 6
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "sympy/sympy": {
          "resolved": 28,
          "total": 77
        },
        "sphinx-doc/sphinx": {
          "resolved": 3,
          "total": 16
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 6
        },
        "pytest-dev/pytest": {
          "resolved": 6,
          "total": 17
        },
        "scikit-learn/scikit-learn": {
          "resolved": 13,
          "total": 23
        },
        "psf/requests": {
          "resolved": 3,
          "total": 6
        },
        "matplotlib/matplotlib": {
          "resolved": 6,
          "total": 23
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 4
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240702_codestory_aide_mixed"
    },
    {
      "id": "lite/20240630_agentless_gpt4o",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-06-30",
      "harness": "agentless",
      "modelSlug": "gpt4o",
      "model": "Gpt4o",
      "resolved": 82,
      "attempted": 292,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 27.3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sphinx-doc/sphinx": {
          "resolved": 3,
          "total": 16
        },
        "django/django": {
          "resolved": 37,
          "total": 114
        },
        "sympy/sympy": {
          "resolved": 14,
          "total": 77
        },
        "astropy/astropy": {
          "resolved": 1,
          "total": 6
        },
        "scikit-learn/scikit-learn": {
          "resolved": 8,
          "total": 23
        },
        "matplotlib/matplotlib": {
          "resolved": 5,
          "total": 23
        },
        "pytest-dev/pytest": {
          "resolved": 6,
          "total": 17
        },
        "psf/requests": {
          "resolved": 4,
          "total": 6
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 6
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240630_agentless_gpt4o"
    },
    {
      "id": "lite/20240627_abanteai_mentatbot_gpt4o",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-06-27",
      "harness": "abanteai",
      "modelSlug": "mentatbot_gpt4o",
      "model": "Mentatbot_gpt4o",
      "resolved": 114,
      "attempted": 296,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 38,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "django/django": {
          "resolved": 44,
          "total": 114
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "sympy/sympy": {
          "resolved": 20,
          "total": 77
        },
        "pytest-dev/pytest": {
          "resolved": 7,
          "total": 17
        },
        "scikit-learn/scikit-learn": {
          "resolved": 16,
          "total": 23
        },
        "sphinx-doc/sphinx": {
          "resolved": 5,
          "total": 16
        },
        "matplotlib/matplotlib": {
          "resolved": 11,
          "total": 23
        },
        "psf/requests": {
          "resolved": 4,
          "total": 6
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 6
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240627_abanteai_mentatbot_gpt4o"
    },
    {
      "id": "lite/20240623_moatless_claude35sonnet",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-06-23",
      "harness": "moatless",
      "modelSlug": "claude35sonnet",
      "model": "Claude35sonnet",
      "resolved": 80,
      "attempted": 294,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 26.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pytest-dev/pytest": {
          "resolved": 3,
          "total": 17
        },
        "sympy/sympy": {
          "resolved": 14,
          "total": 77
        },
        "django/django": {
          "resolved": 41,
          "total": 114
        },
        "sphinx-doc/sphinx": {
          "resolved": 2,
          "total": 16
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 4
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "matplotlib/matplotlib": {
          "resolved": 5,
          "total": 23
        },
        "scikit-learn/scikit-learn": {
          "resolved": 8,
          "total": 23
        },
        "psf/requests": {
          "resolved": 3,
          "total": 6
        },
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 6
        },
        "astropy/astropy": {
          "resolved": 1,
          "total": 6
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240623_moatless_claude35sonnet"
    },
    {
      "id": "lite/20240622_Lingma_Agent",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-06-22",
      "harness": "Lingma",
      "modelSlug": "Agent",
      "model": "Agent",
      "resolved": 99,
      "attempted": 292,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 33,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 6
        },
        "django/django": {
          "resolved": 41,
          "total": 114
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "scikit-learn/scikit-learn": {
          "resolved": 11,
          "total": 23
        },
        "sympy/sympy": {
          "resolved": 19,
          "total": 77
        },
        "matplotlib/matplotlib": {
          "resolved": 7,
          "total": 23
        },
        "sphinx-doc/sphinx": {
          "resolved": 3,
          "total": 16
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 4
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "psf/requests": {
          "resolved": 5,
          "total": 6
        },
        "pytest-dev/pytest": {
          "resolved": 6,
          "total": 17
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240622_Lingma_Agent"
    },
    {
      "id": "lite/20240620_sweagent_claude3.5sonnet",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-06-20",
      "harness": "sweagent",
      "modelSlug": "claude3.5sonnet",
      "model": "Claude3.5sonnet",
      "resolved": 69,
      "attempted": 81,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 23,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "matplotlib/matplotlib": {
          "resolved": 5,
          "total": 23
        },
        "django/django": {
          "resolved": 33,
          "total": 114
        },
        "sphinx-doc/sphinx": {
          "resolved": 0,
          "total": 16
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "sympy/sympy": {
          "resolved": 15,
          "total": 77
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 4
        },
        "pytest-dev/pytest": {
          "resolved": 4,
          "total": 17
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "astropy/astropy": {
          "resolved": 1,
          "total": 6
        },
        "psf/requests": {
          "resolved": 1,
          "total": 6
        },
        "scikit-learn/scikit-learn": {
          "resolved": 9,
          "total": 23
        },
        "pylint-dev/pylint": {
          "resolved": 0,
          "total": 6
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240620_sweagent_claude3.5sonnet"
    },
    {
      "id": "lite/20240617_factory_code_droid",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-06-17",
      "harness": "factory",
      "modelSlug": "code_droid",
      "model": "Code_droid",
      "resolved": 94,
      "attempted": 299,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 31.3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "django/django": {
          "resolved": 42,
          "total": 114
        },
        "matplotlib/matplotlib": {
          "resolved": 5,
          "total": 23
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "psf/requests": {
          "resolved": 3,
          "total": 6
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 6
        },
        "pytest-dev/pytest": {
          "resolved": 4,
          "total": 17
        },
        "scikit-learn/scikit-learn": {
          "resolved": 7,
          "total": 23
        },
        "sphinx-doc/sphinx": {
          "resolved": 6,
          "total": 16
        },
        "sympy/sympy": {
          "resolved": 20,
          "total": 77
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240617_factory_code_droid"
    },
    {
      "id": "lite/20240617_moatless_gpt4o",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-06-17",
      "harness": "moatless",
      "modelSlug": "gpt4o",
      "model": "Gpt4o",
      "resolved": 74,
      "attempted": 289,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 24.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sympy/sympy": {
          "resolved": 15,
          "total": 77
        },
        "pytest-dev/pytest": {
          "resolved": 4,
          "total": 17
        },
        "django/django": {
          "resolved": 36,
          "total": 114
        },
        "matplotlib/matplotlib": {
          "resolved": 3,
          "total": 23
        },
        "sphinx-doc/sphinx": {
          "resolved": 2,
          "total": 16
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "scikit-learn/scikit-learn": {
          "resolved": 7,
          "total": 23
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "astropy/astropy": {
          "resolved": 1,
          "total": 6
        },
        "psf/requests": {
          "resolved": 3,
          "total": 6
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240617_moatless_gpt4o"
    },
    {
      "id": "lite/20240615_appmap-navie_gpt4o",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-06-15",
      "harness": "appmap-navie",
      "modelSlug": "gpt4o",
      "model": "Gpt4o",
      "resolved": 65,
      "attempted": 290,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 21.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "django/django": {
          "resolved": 37,
          "total": 114
        },
        "matplotlib/matplotlib": {
          "resolved": 5,
          "total": 23
        },
        "psf/requests": {
          "resolved": 2,
          "total": 6
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "pytest-dev/pytest": {
          "resolved": 3,
          "total": 17
        },
        "sympy/sympy": {
          "resolved": 6,
          "total": 77
        },
        "scikit-learn/scikit-learn": {
          "resolved": 7,
          "total": 23
        },
        "sphinx-doc/sphinx": {
          "resolved": 1,
          "total": 16
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240615_appmap-navie_gpt4o"
    },
    {
      "id": "lite/20240612_MASAI_gpt4o",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-06-12",
      "harness": "MASAI",
      "modelSlug": "gpt4o",
      "model": "Gpt4o",
      "resolved": 82,
      "attempted": 102,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 27.3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pytest-dev/pytest": {
          "resolved": 5,
          "total": 17
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 3
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "scikit-learn/scikit-learn": {
          "resolved": 9,
          "total": 23
        },
        "sphinx-doc/sphinx": {
          "resolved": 3,
          "total": 16
        },
        "django/django": {
          "resolved": 38,
          "total": 114
        },
        "psf/requests": {
          "resolved": 3,
          "total": 6
        },
        "pylint-dev/pylint": {
          "resolved": 0,
          "total": 6
        },
        "matplotlib/matplotlib": {
          "resolved": 2,
          "total": 23
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "sympy/sympy": {
          "resolved": 17,
          "total": 77
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240612_MASAI_gpt4o"
    },
    {
      "id": "lite/20240612_IBM_Research_Agent101",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-06-12",
      "harness": "IBM",
      "modelSlug": "Research_Agent101",
      "model": "Research_Agent101",
      "resolved": 80,
      "attempted": 293,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 26.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sympy/sympy": {
          "resolved": 14,
          "total": 77
        },
        "django/django": {
          "resolved": 42,
          "total": 114
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "scikit-learn/scikit-learn": {
          "resolved": 7,
          "total": 23
        },
        "matplotlib/matplotlib": {
          "resolved": 4,
          "total": 23
        },
        "pytest-dev/pytest": {
          "resolved": 4,
          "total": 17
        },
        "psf/requests": {
          "resolved": 3,
          "total": 6
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "sphinx-doc/sphinx": {
          "resolved": 1,
          "total": 16
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240612_IBM_Research_Agent101"
    },
    {
      "id": "lite/20240524_opencsg_starship_gpt4",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-05-24",
      "harness": "opencsg",
      "modelSlug": "starship_gpt4",
      "model": "Starship_gpt4",
      "resolved": 71,
      "attempted": 299,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 23.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "astropy/astropy": {
          "resolved": 1,
          "total": 6
        },
        "django/django": {
          "resolved": 36,
          "total": 114
        },
        "matplotlib/matplotlib": {
          "resolved": 1,
          "total": 23
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 4
        },
        "psf/requests": {
          "resolved": 4,
          "total": 6
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "pytest-dev/pytest": {
          "resolved": 5,
          "total": 17
        },
        "scikit-learn/scikit-learn": {
          "resolved": 4,
          "total": 23
        },
        "sphinx-doc/sphinx": {
          "resolved": 4,
          "total": 16
        },
        "sympy/sympy": {
          "resolved": 13,
          "total": 77
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240524_opencsg_starship_gpt4"
    },
    {
      "id": "lite/20240402_sweagent_gpt4",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-04-02",
      "harness": "sweagent",
      "modelSlug": "gpt4",
      "model": "Gpt4",
      "resolved": 54,
      "attempted": 284,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 18,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "astropy/astropy": {
          "resolved": 1,
          "total": 6
        },
        "pytest-dev/pytest": {
          "resolved": 3,
          "total": 17
        },
        "psf/requests": {
          "resolved": 2,
          "total": 6
        },
        "django/django": {
          "resolved": 30,
          "total": 114
        },
        "scikit-learn/scikit-learn": {
          "resolved": 4,
          "total": 23
        },
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 6
        },
        "sphinx-doc/sphinx": {
          "resolved": 1,
          "total": 16
        },
        "sympy/sympy": {
          "resolved": 8,
          "total": 77
        },
        "matplotlib/matplotlib": {
          "resolved": 3,
          "total": 23
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 4
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240402_sweagent_gpt4"
    },
    {
      "id": "lite/20240402_sweagent_claude3opus",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-04-02",
      "harness": "sweagent",
      "modelSlug": "claude3opus",
      "model": "Claude3opus",
      "resolved": 35,
      "attempted": 271,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 11.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "django/django": {
          "resolved": 19,
          "total": 114
        },
        "scikit-learn/scikit-learn": {
          "resolved": 4,
          "total": 23
        },
        "sphinx-doc/sphinx": {
          "resolved": 1,
          "total": 16
        },
        "sympy/sympy": {
          "resolved": 4,
          "total": 77
        },
        "pytest-dev/pytest": {
          "resolved": 1,
          "total": 17
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 6
        },
        "psf/requests": {
          "resolved": 1,
          "total": 6
        },
        "matplotlib/matplotlib": {
          "resolved": 3,
          "total": 23
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240402_sweagent_claude3opus"
    },
    {
      "id": "lite/20240402_rag_claude3opus",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-04-02",
      "harness": "rag",
      "modelSlug": "claude3opus",
      "model": "Claude3opus",
      "resolved": 13,
      "attempted": 300,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 4.3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sympy/sympy": {
          "resolved": 2,
          "total": 77
        },
        "django/django": {
          "resolved": 7,
          "total": 114
        },
        "pytest-dev/pytest": {
          "resolved": 1,
          "total": 17
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 4
        },
        "scikit-learn/scikit-learn": {
          "resolved": 1,
          "total": 23
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240402_rag_claude3opus"
    },
    {
      "id": "lite/20240402_rag_gpt4",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2024-04-02",
      "harness": "rag",
      "modelSlug": "gpt4",
      "model": "Gpt4",
      "resolved": 8,
      "attempted": 300,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 2.7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sympy/sympy": {
          "resolved": 1,
          "total": 77
        },
        "django/django": {
          "resolved": 5,
          "total": 114
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 5
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 4
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20240402_rag_gpt4"
    },
    {
      "id": "lite/20231010_rag_claude2",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2023-10-10",
      "harness": "rag",
      "modelSlug": "claude2",
      "model": "Claude2",
      "resolved": 9,
      "attempted": 299,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pytest-dev/pytest": {
          "resolved": 1,
          "total": 17
        },
        "django/django": {
          "resolved": 6,
          "total": 114
        },
        "scikit-learn/scikit-learn": {
          "resolved": 2,
          "total": 23
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20231010_rag_claude2"
    },
    {
      "id": "lite/20231010_rag_swellama7b",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2023-10-10",
      "harness": "rag",
      "modelSlug": "swellama7b",
      "model": "Swellama7b",
      "resolved": 4,
      "attempted": 292,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 1.3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "django/django": {
          "resolved": 2,
          "total": 114
        },
        "psf/requests": {
          "resolved": 1,
          "total": 6
        },
        "sympy/sympy": {
          "resolved": 1,
          "total": 77
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20231010_rag_swellama7b"
    },
    {
      "id": "lite/20231010_rag_swellama13b",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2023-10-10",
      "harness": "rag",
      "modelSlug": "swellama13b",
      "model": "Swellama13b",
      "resolved": 3,
      "attempted": 287,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 1,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "django/django": {
          "resolved": 1,
          "total": 114
        },
        "sympy/sympy": {
          "resolved": 2,
          "total": 77
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20231010_rag_swellama13b"
    },
    {
      "id": "lite/20231010_rag_gpt35",
      "suite": "SWE-bench Lite",
      "split": "lite",
      "date": "2023-10-10",
      "harness": "rag",
      "modelSlug": "gpt35",
      "model": "Gpt35",
      "resolved": 1,
      "attempted": 300,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 0.3,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "django/django": {
          "resolved": 1,
          "total": 114
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/lite/20231010_rag_gpt35"
    },
    {
      "id": "multilingual/20260902_mini-v2.4.6_gemini-3-5-flash",
      "suite": "SWE-bench Multilingual",
      "split": "multilingual",
      "date": "2026-09-02",
      "harness": "mini-v2.4.6",
      "modelSlug": "gemini-3-5-flash",
      "model": "Gemini 3.5 Flash",
      "resolved": 201,
      "attempted": 264,
      "total": 300,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 67,
      "totalCostUsd": 188.11,
      "meanCostUsd": 0.627,
      "totalApiCalls": 13993,
      "byRepo": {
        "apache/druid": {
          "resolved": 3,
          "total": 5
        },
        "apache/lucene": {
          "resolved": 7,
          "total": 9
        },
        "astral-sh/ruff": {
          "resolved": 0,
          "total": 7
        },
        "axios/axios": {
          "resolved": 1,
          "total": 6
        },
        "babel/babel": {
          "resolved": 3,
          "total": 5
        },
        "briannesbitt/carbon": {
          "resolved": 8,
          "total": 10
        },
        "burntsushi/ripgrep": {
          "resolved": 2,
          "total": 2
        },
        "caddyserver/caddy": {
          "resolved": 8,
          "total": 14
        },
        "facebook/docusaurus": {
          "resolved": 1,
          "total": 5
        },
        "faker-ruby/faker": {
          "resolved": 2,
          "total": 2
        },
        "fastlane/fastlane": {
          "resolved": 5,
          "total": 7
        },
        "fluent/fluentd": {
          "resolved": 8,
          "total": 12
        },
        "fmtlib/fmt": {
          "resolved": 7,
          "total": 11
        },
        "gin-gonic/gin": {
          "resolved": 5,
          "total": 8
        },
        "gohugoio/hugo": {
          "resolved": 6,
          "total": 7
        },
        "google/gson": {
          "resolved": 7,
          "total": 9
        },
        "hashicorp/terraform": {
          "resolved": 3,
          "total": 5
        },
        "immutable-js/immutable-js": {
          "resolved": 2,
          "total": 2
        },
        "javaparser/javaparser": {
          "resolved": 2,
          "total": 2
        },
        "jekyll/jekyll": {
          "resolved": 2,
          "total": 5
        },
        "jordansissel/fpm": {
          "resolved": 1,
          "total": 2
        },
        "jqlang/jq": {
          "resolved": 7,
          "total": 9
        },
        "laravel/framework": {
          "resolved": 13,
          "total": 13
        },
        "micropython/micropython": {
          "resolved": 5,
          "total": 5
        },
        "mrdoob/three.js": {
          "resolved": 1,
          "total": 3
        },
        "nlohmann/json": {
          "resolved": 1,
          "total": 1
        },
        "nushell/nushell": {
          "resolved": 0,
          "total": 5
        },
        "php-cs-fixer/php-cs-fixer": {
          "resolved": 5,
          "total": 10
        },
        "phpoffice/phpspreadsheet": {
          "resolved": 7,
          "total": 10
        },
        "preactjs/preact": {
          "resolved": 17,
          "total": 17
        },
        "projectlombok/lombok": {
          "resolved": 11,
          "total": 17
        },
        "prometheus/prometheus": {
          "resolved": 4,
          "total": 8
        },
        "reactivex/rxjava": {
          "resolved": 1,
          "total": 1
        },
        "redis/redis": {
          "resolved": 9,
          "total": 12
        },
        "rubocop/rubocop": {
          "resolved": 15,
          "total": 16
        },
        "sharkdp/bat": {
          "resolved": 7,
          "total": 8
        },
        "tokio-rs/axum": {
          "resolved": 6,
          "total": 7
        },
        "tokio-rs/tokio": {
          "resolved": 9,
          "total": 9
        },
        "uutils/coreutils": {
          "resolved": 0,
          "total": 5
        },
        "valkey-io/valkey": {
          "resolved": 0,
          "total": 4
        },
        "vuejs/core": {
          "resolved": 0,
          "total": 5
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/multilingual/20260902_mini-v2.4.6_gemini-3-5-flash"
    },
    {
      "id": "multilingual/20260220_mini-v2.0.0_gpt-5-2-codex",
      "suite": "SWE-bench Multilingual",
      "split": "multilingual",
      "date": "2026-02-20",
      "harness": "mini-v2.0.0",
      "modelSlug": "gpt-5-2-codex",
      "model": "GPT 5.2 Codex",
      "resolved": 199,
      "attempted": 300,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 66.3,
      "totalCostUsd": 198.65,
      "meanCostUsd": 0.662,
      "totalApiCalls": 11453,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/multilingual/20260220_mini-v2.0.0_gpt-5-2-codex"
    },
    {
      "id": "multilingual/20260216_mini-v2.0.0a0_minimax-2-5",
      "suite": "SWE-bench Multilingual",
      "split": "multilingual",
      "date": "2026-02-16",
      "harness": "mini-v2.0.0a0",
      "modelSlug": "minimax-2-5",
      "model": "Minimax 2.5",
      "resolved": 203,
      "attempted": 297,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 67.7,
      "totalCostUsd": 30.06,
      "meanCostUsd": 0.1,
      "totalApiCalls": 22027,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/multilingual/20260216_mini-v2.0.0a0_minimax-2-5"
    },
    {
      "id": "multilingual/20260213_mini-v2.0.0a0_gemini-3-flash",
      "suite": "SWE-bench Multilingual",
      "split": "multilingual",
      "date": "2026-02-13",
      "harness": "mini-v2.0.0a0",
      "modelSlug": "gemini-3-flash",
      "model": "Gemini 3 Flash",
      "resolved": 218,
      "attempted": 300,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 72.7,
      "totalCostUsd": 105.55,
      "meanCostUsd": 0.352,
      "totalApiCalls": 15736,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/multilingual/20260213_mini-v2.0.0a0_gemini-3-flash"
    },
    {
      "id": "multilingual/20260213_mini-v2.0.0a0_claude-4-6-opus",
      "suite": "SWE-bench Multilingual",
      "split": "multilingual",
      "date": "2026-02-13",
      "harness": "mini-v2.0.0a0",
      "modelSlug": "claude-4-6-opus",
      "model": "Claude 4.6 Opus",
      "resolved": 216,
      "attempted": 300,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 72,
      "totalCostUsd": 198.89,
      "meanCostUsd": 0.663,
      "totalApiCalls": 8656,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/multilingual/20260213_mini-v2.0.0a0_claude-4-6-opus"
    },
    {
      "id": "multilingual/20260213_mini-v2.0.0a0_claude-4-5-opus",
      "suite": "SWE-bench Multilingual",
      "split": "multilingual",
      "date": "2026-02-13",
      "harness": "mini-v2.0.0a0",
      "modelSlug": "claude-4-5-opus",
      "model": "Claude 4.5 Opus",
      "resolved": 212,
      "attempted": 300,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 70.7,
      "totalCostUsd": 250.07,
      "meanCostUsd": 0.834,
      "totalApiCalls": 9239,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/multilingual/20260213_mini-v2.0.0a0_claude-4-5-opus"
    },
    {
      "id": "multilingual/20260213_mini-v2.0.0a0_glm-5",
      "suite": "SWE-bench Multilingual",
      "split": "multilingual",
      "date": "2026-02-13",
      "harness": "mini-v2.0.0a0",
      "modelSlug": "glm-5",
      "model": "GLM 5",
      "resolved": 209,
      "attempted": 300,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 69.7,
      "totalCostUsd": 192.74,
      "meanCostUsd": 0.642,
      "totalApiCalls": 20269,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/multilingual/20260213_mini-v2.0.0a0_glm-5"
    },
    {
      "id": "multilingual/20260213_mini-v2.0.0a0_gemini-3-pro",
      "suite": "SWE-bench Multilingual",
      "split": "multilingual",
      "date": "2026-02-13",
      "harness": "mini-v2.0.0a0",
      "modelSlug": "gemini-3-pro",
      "model": "Gemini 3 Pro",
      "resolved": 206,
      "attempted": 300,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 68.7,
      "totalCostUsd": 306.33,
      "meanCostUsd": 1.021,
      "totalApiCalls": 14655,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/multilingual/20260213_mini-v2.0.0a0_gemini-3-pro"
    },
    {
      "id": "multilingual/20260213_mini-v2.0.0a0_kimi-k2-5",
      "suite": "SWE-bench Multilingual",
      "split": "multilingual",
      "date": "2026-02-13",
      "harness": "mini-v2.0.0a0",
      "modelSlug": "kimi-k2-5",
      "model": "Kimi K2 5",
      "resolved": 202,
      "attempted": 300,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 67.3,
      "totalCostUsd": 207.85,
      "meanCostUsd": 0.693,
      "totalApiCalls": 15130,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/multilingual/20260213_mini-v2.0.0a0_kimi-k2-5"
    },
    {
      "id": "multilingual/20260213_mini-v2.0.0a0_claude-4-5-sonnet",
      "suite": "SWE-bench Multilingual",
      "split": "multilingual",
      "date": "2026-02-13",
      "harness": "mini-v2.0.0a0",
      "modelSlug": "claude-4-5-sonnet",
      "model": "Claude 4.5 Sonnet",
      "resolved": 201,
      "attempted": 300,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 67,
      "totalCostUsd": 200.86,
      "meanCostUsd": 0.67,
      "totalApiCalls": 12763,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/multilingual/20260213_mini-v2.0.0a0_claude-4-5-sonnet"
    },
    {
      "id": "multilingual/20260213_mini-v2.0.0a0_gpt-5-2-high",
      "suite": "SWE-bench Multilingual",
      "split": "multilingual",
      "date": "2026-02-13",
      "harness": "mini-v2.0.0a0",
      "modelSlug": "gpt-5-2-high",
      "model": "GPT 5.2",
      "resolved": 200,
      "attempted": 300,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 66.7,
      "totalCostUsd": 160.92,
      "meanCostUsd": 0.536,
      "totalApiCalls": 12080,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/multilingual/20260213_mini-v2.0.0a0_gpt-5-2-high"
    },
    {
      "id": "multilingual/20260213_mini-v2.0.0a0_claude-4-5-haiku",
      "suite": "SWE-bench Multilingual",
      "split": "multilingual",
      "date": "2026-02-13",
      "harness": "mini-v2.0.0a0",
      "modelSlug": "claude-4-5-haiku",
      "model": "Claude 4.5 Haiku",
      "resolved": 194,
      "attempted": 300,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 64.7,
      "totalCostUsd": 113.85,
      "meanCostUsd": 0.38,
      "totalApiCalls": 21043,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/multilingual/20260213_mini-v2.0.0a0_claude-4-5-haiku"
    },
    {
      "id": "multilingual/20260213_mini-v2.0.0a0_deepseek-3-2",
      "suite": "SWE-bench Multilingual",
      "split": "multilingual",
      "date": "2026-02-13",
      "harness": "mini-v2.0.0a0",
      "modelSlug": "deepseek-3-2",
      "model": "DeepSeek 3.2",
      "resolved": 177,
      "attempted": 300,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 59,
      "totalCostUsd": 115.15,
      "meanCostUsd": 0.384,
      "totalApiCalls": 24797,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/multilingual/20260213_mini-v2.0.0a0_deepseek-3-2"
    },
    {
      "id": "multilingual/20260213_mini-v2.0.0a0_gpt-5-mini",
      "suite": "SWE-bench Multilingual",
      "split": "multilingual",
      "date": "2026-02-13",
      "harness": "mini-v2.0.0a0",
      "modelSlug": "gpt-5-mini",
      "model": "GPT 5 Mini",
      "resolved": 119,
      "attempted": 300,
      "total": 300,
      "totalBasis": "split-constant",
      "resolvedPct": 39.7,
      "totalCostUsd": 15.49,
      "meanCostUsd": 0.052,
      "totalApiCalls": 9000,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/multilingual/20260213_mini-v2.0.0a0_gpt-5-mini"
    },
    {
      "id": "verified/20260901_mini-v2.4.2_gemini-3-5-flash",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2026-09-01",
      "harness": "mini-v2.4.2",
      "modelSlug": "gemini-3-5-flash",
      "model": "Gemini 3.5 Flash",
      "resolved": 359,
      "attempted": 441,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 71.8,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": 19515,
      "byRepo": {
        "astropy/astropy": {
          "resolved": 18,
          "total": 22
        },
        "django/django": {
          "resolved": 187,
          "total": 231
        },
        "matplotlib/matplotlib": {
          "resolved": 1,
          "total": 34
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "psf/requests": {
          "resolved": 8,
          "total": 8
        },
        "pydata/xarray": {
          "resolved": 12,
          "total": 22
        },
        "pylint-dev/pylint": {
          "resolved": 0,
          "total": 10
        },
        "pytest-dev/pytest": {
          "resolved": 16,
          "total": 19
        },
        "scikit-learn/scikit-learn": {
          "resolved": 26,
          "total": 32
        },
        "sphinx-doc/sphinx": {
          "resolved": 33,
          "total": 44
        },
        "sympy/sympy": {
          "resolved": 57,
          "total": 75
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20260901_mini-v2.4.2_gemini-3-5-flash"
    },
    {
      "id": "verified/20260217_mini-v2.0.0_claude-4-5-opus-high",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2026-02-17",
      "harness": "mini-v2.0.0",
      "modelSlug": "claude-4-5-opus-high",
      "model": "Claude 4.5 Opus",
      "resolved": 384,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 76.8,
      "totalCostUsd": 376.95,
      "meanCostUsd": 0.754,
      "totalApiCalls": 16448,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20260217_mini-v2.0.0_claude-4-5-opus-high"
    },
    {
      "id": "verified/20260217_mini-v2.0.0_gemini-3-flash-high",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2026-02-17",
      "harness": "mini-v2.0.0",
      "modelSlug": "gemini-3-flash-high",
      "model": "Gemini 3 Flash",
      "resolved": 379,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 75.8,
      "totalCostUsd": 177.98,
      "meanCostUsd": 0.356,
      "totalApiCalls": 28063,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20260217_mini-v2.0.0_gemini-3-flash-high"
    },
    {
      "id": "verified/20260217_mini-v2.0.0_minimax-2-5-high",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2026-02-17",
      "harness": "mini-v2.0.0",
      "modelSlug": "minimax-2-5-high",
      "model": "Minimax 2.5",
      "resolved": 379,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 75.8,
      "totalCostUsd": 36.64,
      "meanCostUsd": 0.073,
      "totalApiCalls": 30225,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20260217_mini-v2.0.0_minimax-2-5-high"
    },
    {
      "id": "verified/20260217_mini-v2.0.0_claude-4-6-opus",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2026-02-17",
      "harness": "mini-v2.0.0",
      "modelSlug": "claude-4-6-opus",
      "model": "Claude 4.6 Opus",
      "resolved": 378,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 75.6,
      "totalCostUsd": 275.76,
      "meanCostUsd": 0.552,
      "totalApiCalls": 14466,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20260217_mini-v2.0.0_claude-4-6-opus"
    },
    {
      "id": "verified/20260217_mini-v2.0.0_glm-5-high",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2026-02-17",
      "harness": "mini-v2.0.0",
      "modelSlug": "glm-5-high",
      "model": "GLM 5",
      "resolved": 364,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 72.8,
      "totalCostUsd": 267.19,
      "meanCostUsd": 0.534,
      "totalApiCalls": 38088,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20260217_mini-v2.0.0_glm-5-high"
    },
    {
      "id": "verified/20260217_mini-v2.0.0_gpt-5-2-high",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2026-02-17",
      "harness": "mini-v2.0.0",
      "modelSlug": "gpt-5-2-high",
      "model": "GPT 5.2",
      "resolved": 364,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 72.8,
      "totalCostUsd": 236.78,
      "meanCostUsd": 0.474,
      "totalApiCalls": 17523,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20260217_mini-v2.0.0_gpt-5-2-high"
    },
    {
      "id": "verified/20260217_mini-v2.0.0_claude-4-5-sonnet-high",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2026-02-17",
      "harness": "mini-v2.0.0",
      "modelSlug": "claude-4-5-sonnet-high",
      "model": "Claude 4.5 Sonnet",
      "resolved": 357,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 71.4,
      "totalCostUsd": 328.95,
      "meanCostUsd": 0.658,
      "totalApiCalls": 24150,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20260217_mini-v2.0.0_claude-4-5-sonnet-high"
    },
    {
      "id": "verified/20260217_mini-v2.0.0_kimi-k2-5-high",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2026-02-17",
      "harness": "mini-v2.0.0",
      "modelSlug": "kimi-k2-5-high",
      "model": "Kimi K2 5",
      "resolved": 354,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 70.8,
      "totalCostUsd": 73.28,
      "meanCostUsd": 0.147,
      "totalApiCalls": 25589,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20260217_mini-v2.0.0_kimi-k2-5-high"
    },
    {
      "id": "verified/20260217_mini-v2.0.0_deepseek-3-2-high",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2026-02-17",
      "harness": "mini-v2.0.0",
      "modelSlug": "deepseek-3-2-high",
      "model": "DeepSeek 3.2",
      "resolved": 350,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 70,
      "totalCostUsd": 223.92,
      "meanCostUsd": 0.448,
      "totalApiCalls": 44253,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20260217_mini-v2.0.0_deepseek-3-2-high"
    },
    {
      "id": "verified/20260217_mini-v2.0.0_claude-4-5-haiku-high",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2026-02-17",
      "harness": "mini-v2.0.0",
      "modelSlug": "claude-4-5-haiku-high",
      "model": "Claude 4.5 Haiku",
      "resolved": 333,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 66.6,
      "totalCostUsd": 165.46,
      "meanCostUsd": 0.331,
      "totalApiCalls": 33077,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20260217_mini-v2.0.0_claude-4-5-haiku-high"
    },
    {
      "id": "verified/20260217_mini-v2.0.0_gpt-5-mini",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2026-02-17",
      "harness": "mini-v2.0.0",
      "modelSlug": "gpt-5-mini",
      "model": "GPT 5 Mini",
      "resolved": 281,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 56.2,
      "totalCostUsd": 23.6,
      "meanCostUsd": 0.047,
      "totalApiCalls": 10171,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20260217_mini-v2.0.0_gpt-5-mini"
    },
    {
      "id": "verified/20251215_livesweagent_claude-opus-4-5",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-12-15",
      "harness": "livesweagent",
      "modelSlug": "claude-opus-4-5",
      "model": "Claude Opus 4.5",
      "resolved": 396,
      "attempted": 401,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 79.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "psf/requests": {
          "resolved": 7,
          "total": 8
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "scikit-learn/scikit-learn": {
          "resolved": 30,
          "total": 32
        },
        "django/django": {
          "resolved": 190,
          "total": 231
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "astropy/astropy": {
          "resolved": 11,
          "total": 22
        },
        "sphinx-doc/sphinx": {
          "resolved": 32,
          "total": 44
        },
        "sympy/sympy": {
          "resolved": 57,
          "total": 75
        },
        "matplotlib/matplotlib": {
          "resolved": 25,
          "total": 34
        },
        "pylint-dev/pylint": {
          "resolved": 6,
          "total": 10
        },
        "pydata/xarray": {
          "resolved": 19,
          "total": 22
        },
        "pytest-dev/pytest": {
          "resolved": 17,
          "total": 19
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251215_livesweagent_claude-opus-4-5"
    },
    {
      "id": "verified/20251211_mini-v1.17.2_gpt-5.2-2025-12-11-high",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-12-11",
      "harness": "mini-v1.17.2",
      "modelSlug": "gpt-5.2-2025-12-11-high",
      "model": "GPT 5.2 2025 12 11",
      "resolved": 359,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 71.8,
      "totalCostUsd": 260.13,
      "meanCostUsd": 0.52,
      "totalApiCalls": 9881,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251211_mini-v1.17.2_gpt-5.2-2025-12-11-high"
    },
    {
      "id": "verified/20251211_mini-v1.17.2_gpt-5.2-2025-12-11",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-12-11",
      "harness": "mini-v1.17.2",
      "modelSlug": "gpt-5.2-2025-12-11",
      "model": "GPT 5.2 2025 12 11",
      "resolved": 345,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 69,
      "totalCostUsd": 134.83,
      "meanCostUsd": 0.27,
      "totalApiCalls": 8135,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251211_mini-v1.17.2_gpt-5.2-2025-12-11"
    },
    {
      "id": "verified/20251210_mini-v1.17.2_kimi-k2-thinking",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-12-10",
      "harness": "mini-v1.17.2",
      "modelSlug": "kimi-k2-thinking",
      "model": "Kimi K2",
      "resolved": 317,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 63.4,
      "totalCostUsd": 219.15,
      "meanCostUsd": 0.438,
      "totalApiCalls": 23410,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251210_mini-v1.17.2_kimi-k2-thinking"
    },
    {
      "id": "verified/20251209_mini-v1.17.2_devstral-small-2512",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-12-09",
      "harness": "mini-v1.17.2",
      "modelSlug": "devstral-small-2512",
      "model": "Devstral Small 2512",
      "resolved": 282,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 56.4,
      "totalCostUsd": 119.02,
      "meanCostUsd": 0.238,
      "totalApiCalls": 43433,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251209_mini-v1.17.2_devstral-small-2512"
    },
    {
      "id": "verified/20251209_mini-v1.17.2_devstral-2512",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-12-09",
      "harness": "mini-v1.17.2",
      "modelSlug": "devstral-2512",
      "model": "Devstral 2512",
      "resolved": 269,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 53.8,
      "totalCostUsd": 338,
      "meanCostUsd": 0.676,
      "totalApiCalls": 37543,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251209_mini-v1.17.2_devstral-2512"
    },
    {
      "id": "verified/20251205_sonar-foundation-agent_claude-opus-4-5",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-12-05",
      "harness": "sonar-foundation-agent",
      "modelSlug": "claude-opus-4-5",
      "model": "Claude Opus 4.5",
      "resolved": 396,
      "attempted": 396,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 79.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 10
        },
        "matplotlib/matplotlib": {
          "resolved": 27,
          "total": 34
        },
        "sympy/sympy": {
          "resolved": 58,
          "total": 75
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "pytest-dev/pytest": {
          "resolved": 18,
          "total": 19
        },
        "pydata/xarray": {
          "resolved": 19,
          "total": 22
        },
        "scikit-learn/scikit-learn": {
          "resolved": 30,
          "total": 32
        },
        "django/django": {
          "resolved": 191,
          "total": 231
        },
        "psf/requests": {
          "resolved": 5,
          "total": 8
        },
        "sphinx-doc/sphinx": {
          "resolved": 30,
          "total": 44
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "astropy/astropy": {
          "resolved": 12,
          "total": 22
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251205_sonar-foundation-agent_claude-opus-4-5"
    },
    {
      "id": "verified/20251201_mini-v1.17.1_deepseek-v3.2-reasoner",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-12-01",
      "harness": "mini-v1.17.1",
      "modelSlug": "deepseek-v3.2-reasoner",
      "model": "DeepSeek V3.2 Reasoner",
      "resolved": 300,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 60,
      "totalCostUsd": 14.04,
      "meanCostUsd": 0.028,
      "totalApiCalls": 23202,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251201_mini-v1.17.1_deepseek-v3.2-reasoner"
    },
    {
      "id": "verified/20251201_mini-v1.17.1_glm-4.6",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-12-01",
      "harness": "mini-v1.17.1",
      "modelSlug": "glm-4.6",
      "model": "GLM 4.6",
      "resolved": 277,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 55.4,
      "totalCostUsd": 48.31,
      "meanCostUsd": 0.097,
      "totalApiCalls": 24674,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251201_mini-v1.17.1_glm-4.6"
    },
    {
      "id": "verified/20251127_openhands_claude-opus-4-5",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-11-27",
      "harness": "openhands",
      "modelSlug": "claude-opus-4-5",
      "model": "Claude Opus 4.5",
      "resolved": 388,
      "attempted": 390,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 77.6,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "matplotlib/matplotlib": {
          "resolved": 26,
          "total": 34
        },
        "psf/requests": {
          "resolved": 6,
          "total": 8
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "pydata/xarray": {
          "resolved": 18,
          "total": 22
        },
        "pylint-dev/pylint": {
          "resolved": 5,
          "total": 10
        },
        "astropy/astropy": {
          "resolved": 12,
          "total": 22
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "sympy/sympy": {
          "resolved": 57,
          "total": 75
        },
        "pytest-dev/pytest": {
          "resolved": 17,
          "total": 19
        },
        "sphinx-doc/sphinx": {
          "resolved": 30,
          "total": 44
        },
        "scikit-learn/scikit-learn": {
          "resolved": 30,
          "total": 32
        },
        "django/django": {
          "resolved": 185,
          "total": 231
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251127_openhands_claude-opus-4-5"
    },
    {
      "id": "verified/20251124_mini-v1.16.0_claude-opus-4-5-20251101",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-11-24",
      "harness": "mini-v1.16.0",
      "modelSlug": "claude-opus-4-5-20251101",
      "model": "Claude Opus 4.5 20251101",
      "resolved": 372,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 74.4,
      "totalCostUsd": 360.62,
      "meanCostUsd": 0.721,
      "totalApiCalls": 19021,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251124_mini-v1.16.0_claude-opus-4-5-20251101"
    },
    {
      "id": "verified/20251124_mini-v1.16.0_gpt-5.1-codex",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-11-24",
      "harness": "mini-v1.16.0",
      "modelSlug": "gpt-5.1-codex",
      "model": "GPT 5.1 Codex",
      "resolved": 330,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 66,
      "totalCostUsd": 294.44,
      "meanCostUsd": 0.589,
      "totalApiCalls": 11941,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251124_mini-v1.16.0_gpt-5.1-codex"
    },
    {
      "id": "verified/20251124_mini-v1.17.0_minimax-m2",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-11-24",
      "harness": "mini-v1.17.0",
      "modelSlug": "minimax-m2",
      "model": "Minimax M2",
      "resolved": 305,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 61,
      "totalCostUsd": 214.13,
      "meanCostUsd": 0.428,
      "totalApiCalls": 37069,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251124_mini-v1.17.0_minimax-m2"
    },
    {
      "id": "verified/20251120_livesweagent_gemini-3-pro-preview",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-11-20",
      "harness": "livesweagent",
      "modelSlug": "gemini-3-pro-preview",
      "model": "Gemini 3 Pro",
      "resolved": 387,
      "attempted": 390,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 77.4,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sphinx-doc/sphinx": {
          "resolved": 30,
          "total": 44
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 10
        },
        "sympy/sympy": {
          "resolved": 53,
          "total": 75
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "matplotlib/matplotlib": {
          "resolved": 26,
          "total": 34
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "psf/requests": {
          "resolved": 4,
          "total": 8
        },
        "pytest-dev/pytest": {
          "resolved": 17,
          "total": 19
        },
        "scikit-learn/scikit-learn": {
          "resolved": 29,
          "total": 32
        },
        "pydata/xarray": {
          "resolved": 18,
          "total": 22
        },
        "django/django": {
          "resolved": 191,
          "total": 231
        },
        "astropy/astropy": {
          "resolved": 14,
          "total": 22
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251120_livesweagent_gemini-3-pro-preview"
    },
    {
      "id": "verified/20251120_mini-v1.15.0_gpt-5.1-2025-11-13",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-11-20",
      "harness": "mini-v1.15.0",
      "modelSlug": "gpt-5.1-2025-11-13",
      "model": "GPT 5.1 2025 11 13",
      "resolved": 330,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 66,
      "totalCostUsd": 153.08,
      "meanCostUsd": 0.306,
      "totalApiCalls": 10500,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251120_mini-v1.15.0_gpt-5.1-2025-11-13"
    },
    {
      "id": "verified/20251118_mini-v1.15.0_gemini-3-pro-preview-20251118",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-11-18",
      "harness": "mini-v1.15.0",
      "modelSlug": "gemini-3-pro-preview-20251118",
      "model": "Gemini 3 Pro Preview 20251118",
      "resolved": 371,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 74.2,
      "totalCostUsd": 229.98,
      "meanCostUsd": 0.46,
      "totalApiCalls": 20163,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251118_mini-v1.15.0_gemini-3-pro-preview-20251118"
    },
    {
      "id": "verified/20251103_sonar-foundation-agent_claude-sonnet-4-5",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-11-03",
      "harness": "sonar-foundation-agent",
      "modelSlug": "claude-sonnet-4-5",
      "model": "Claude Sonnet 4.5",
      "resolved": 374,
      "attempted": 377,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 74.8,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "scikit-learn/scikit-learn": {
          "resolved": 28,
          "total": 32
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "matplotlib/matplotlib": {
          "resolved": 26,
          "total": 34
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "sphinx-doc/sphinx": {
          "resolved": 30,
          "total": 44
        },
        "psf/requests": {
          "resolved": 3,
          "total": 8
        },
        "pylint-dev/pylint": {
          "resolved": 5,
          "total": 10
        },
        "pydata/xarray": {
          "resolved": 16,
          "total": 22
        },
        "pytest-dev/pytest": {
          "resolved": 16,
          "total": 19
        },
        "django/django": {
          "resolved": 183,
          "total": 231
        },
        "astropy/astropy": {
          "resolved": 12,
          "total": 22
        },
        "sympy/sympy": {
          "resolved": 53,
          "total": 75
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251103_sonar-foundation-agent_claude-sonnet-4-5"
    },
    {
      "id": "verified/20251103_SalesforceAIResearch_SAGE_OpenHands",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-11-03",
      "harness": "SalesforceAIResearch",
      "modelSlug": "SAGE_OpenHands",
      "model": "SAGE_OpenHands",
      "resolved": 369,
      "attempted": 372,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 73.8,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "astropy/astropy": {
          "resolved": 13,
          "total": 22
        },
        "sympy/sympy": {
          "resolved": 56,
          "total": 75
        },
        "psf/requests": {
          "resolved": 6,
          "total": 8
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 10
        },
        "matplotlib/matplotlib": {
          "resolved": 23,
          "total": 34
        },
        "pydata/xarray": {
          "resolved": 16,
          "total": 22
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "pytest-dev/pytest": {
          "resolved": 14,
          "total": 19
        },
        "sphinx-doc/sphinx": {
          "resolved": 30,
          "total": 44
        },
        "django/django": {
          "resolved": 176,
          "total": 231
        },
        "scikit-learn/scikit-learn": {
          "resolved": 29,
          "total": 32
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251103_SalesforceAIResearch_SAGE_OpenHands"
    },
    {
      "id": "verified/20251021_SalesforceAIResearch_SAGE_bash_only",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-10-21",
      "harness": "SalesforceAIResearch",
      "modelSlug": "SAGE_bash_only",
      "model": "SAGE_bash_only",
      "resolved": 365,
      "attempted": 366,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 73,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "psf/requests": {
          "resolved": 5,
          "total": 8
        },
        "django/django": {
          "resolved": 176,
          "total": 231
        },
        "sphinx-doc/sphinx": {
          "resolved": 31,
          "total": 44
        },
        "pydata/xarray": {
          "resolved": 19,
          "total": 22
        },
        "pylint-dev/pylint": {
          "resolved": 5,
          "total": 10
        },
        "scikit-learn/scikit-learn": {
          "resolved": 28,
          "total": 32
        },
        "matplotlib/matplotlib": {
          "resolved": 18,
          "total": 34
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "astropy/astropy": {
          "resolved": 15,
          "total": 22
        },
        "sympy/sympy": {
          "resolved": 50,
          "total": 75
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "pytest-dev/pytest": {
          "resolved": 16,
          "total": 19
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251021_SalesforceAIResearch_SAGE_bash_only"
    },
    {
      "id": "verified/20251015_Prometheus_v1.2.1_gpt5",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-10-15",
      "harness": "Prometheus",
      "modelSlug": "v1.2.1_gpt5",
      "model": "V1.2.1_gpt5",
      "resolved": 372,
      "attempted": 373,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 74.4,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "matplotlib/matplotlib": {
          "resolved": 23,
          "total": 34
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "pydata/xarray": {
          "resolved": 18,
          "total": 22
        },
        "django/django": {
          "resolved": 179,
          "total": 231
        },
        "pytest-dev/pytest": {
          "resolved": 15,
          "total": 19
        },
        "sympy/sympy": {
          "resolved": 51,
          "total": 75
        },
        "scikit-learn/scikit-learn": {
          "resolved": 30,
          "total": 32
        },
        "psf/requests": {
          "resolved": 4,
          "total": 8
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "pylint-dev/pylint": {
          "resolved": 6,
          "total": 10
        },
        "sphinx-doc/sphinx": {
          "resolved": 32,
          "total": 44
        },
        "astropy/astropy": {
          "resolved": 12,
          "total": 22
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251015_Prometheus_v1.2.1_gpt5"
    },
    {
      "id": "verified/20251014_Lingxi_kimi_k2",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-10-14",
      "harness": "Lingxi",
      "modelSlug": "kimi_k2",
      "model": "Kimi_k2",
      "resolved": 356,
      "attempted": 357,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 71.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pytest-dev/pytest": {
          "resolved": 15,
          "total": 19
        },
        "sympy/sympy": {
          "resolved": 53,
          "total": 75
        },
        "pydata/xarray": {
          "resolved": 19,
          "total": 22
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 10
        },
        "psf/requests": {
          "resolved": 3,
          "total": 8
        },
        "sphinx-doc/sphinx": {
          "resolved": 30,
          "total": 44
        },
        "matplotlib/matplotlib": {
          "resolved": 22,
          "total": 34
        },
        "astropy/astropy": {
          "resolved": 12,
          "total": 22
        },
        "scikit-learn/scikit-learn": {
          "resolved": 28,
          "total": 32
        },
        "django/django": {
          "resolved": 171,
          "total": 231
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20251014_Lingxi_kimi_k2"
    },
    {
      "id": "verified/20250930_zai_glm4-6",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-09-30",
      "harness": "zai",
      "modelSlug": "glm4-6",
      "model": "Glm4 6",
      "resolved": 341,
      "attempted": 344,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 68.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 10
        },
        "psf/requests": {
          "resolved": 4,
          "total": 8
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "pytest-dev/pytest": {
          "resolved": 15,
          "total": 19
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "django/django": {
          "resolved": 163,
          "total": 231
        },
        "scikit-learn/scikit-learn": {
          "resolved": 27,
          "total": 32
        },
        "astropy/astropy": {
          "resolved": 12,
          "total": 22
        },
        "matplotlib/matplotlib": {
          "resolved": 24,
          "total": 34
        },
        "sphinx-doc/sphinx": {
          "resolved": 26,
          "total": 44
        },
        "sympy/sympy": {
          "resolved": 49,
          "total": 75
        },
        "pydata/xarray": {
          "resolved": 16,
          "total": 22
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250930_zai_glm4-6"
    },
    {
      "id": "verified/20250929_Prometheus_v1.2_gpt5",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-09-29",
      "harness": "Prometheus",
      "modelSlug": "v1.2_gpt5",
      "model": "V1.2_gpt5",
      "resolved": 356,
      "attempted": 357,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 71.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sympy/sympy": {
          "resolved": 47,
          "total": 75
        },
        "pydata/xarray": {
          "resolved": 18,
          "total": 22
        },
        "pylint-dev/pylint": {
          "resolved": 5,
          "total": 10
        },
        "django/django": {
          "resolved": 175,
          "total": 231
        },
        "psf/requests": {
          "resolved": 2,
          "total": 8
        },
        "astropy/astropy": {
          "resolved": 12,
          "total": 22
        },
        "sphinx-doc/sphinx": {
          "resolved": 30,
          "total": 44
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "pytest-dev/pytest": {
          "resolved": 14,
          "total": 19
        },
        "scikit-learn/scikit-learn": {
          "resolved": 29,
          "total": 32
        },
        "matplotlib/matplotlib": {
          "resolved": 22,
          "total": 34
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250929_Prometheus_v1.2_gpt5"
    },
    {
      "id": "verified/20250929_mini-v1.13.3_sonnet-4-5-20250929",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-09-29",
      "harness": "mini-v1.13.3",
      "modelSlug": "sonnet-4-5-20250929",
      "model": "Sonnet 4.5 20250929",
      "resolved": 353,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 70.6,
      "totalCostUsd": 279.17,
      "meanCostUsd": 0.558,
      "totalApiCalls": 25494,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250929_mini-v1.13.3_sonnet-4-5-20250929"
    },
    {
      "id": "verified/20250928_trae_doubao_seed_code",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-09-28",
      "harness": "trae",
      "modelSlug": "doubao_seed_code",
      "model": "Doubao_seed_code",
      "resolved": 394,
      "attempted": 396,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 78.8,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "scikit-learn/scikit-learn": {
          "resolved": 30,
          "total": 32
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "matplotlib/matplotlib": {
          "resolved": 27,
          "total": 34
        },
        "pytest-dev/pytest": {
          "resolved": 16,
          "total": 19
        },
        "astropy/astropy": {
          "resolved": 17,
          "total": 22
        },
        "django/django": {
          "resolved": 182,
          "total": 231
        },
        "sympy/sympy": {
          "resolved": 57,
          "total": 75
        },
        "pylint-dev/pylint": {
          "resolved": 6,
          "total": 10
        },
        "psf/requests": {
          "resolved": 6,
          "total": 8
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "sphinx-doc/sphinx": {
          "resolved": 31,
          "total": 44
        },
        "pydata/xarray": {
          "resolved": 20,
          "total": 22
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250928_trae_doubao_seed_code"
    },
    {
      "id": "verified/20250924_artemis_agent_v2",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-09-24",
      "harness": "artemis",
      "modelSlug": "agent_v2",
      "model": "Agent_v2",
      "resolved": 285,
      "attempted": 297,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 57,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sphinx-doc/sphinx": {
          "resolved": 21,
          "total": 44
        },
        "django/django": {
          "resolved": 144,
          "total": 231
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "scikit-learn/scikit-learn": {
          "resolved": 25,
          "total": 32
        },
        "sympy/sympy": {
          "resolved": 41,
          "total": 75
        },
        "psf/requests": {
          "resolved": 1,
          "total": 8
        },
        "astropy/astropy": {
          "resolved": 9,
          "total": 22
        },
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 10
        },
        "pytest-dev/pytest": {
          "resolved": 12,
          "total": 19
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "pydata/xarray": {
          "resolved": 11,
          "total": 22
        },
        "matplotlib/matplotlib": {
          "resolved": 18,
          "total": 34
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250924_artemis_agent_v2"
    },
    {
      "id": "verified/20250901_entroPO_R2E_QwenCoder30BA3B_tts",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-09-01",
      "harness": "entroPO",
      "modelSlug": "R2E_QwenCoder30BA3B_tts",
      "model": "R2E_QwenCoder30BA3B_tts",
      "resolved": 302,
      "attempted": 307,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 60.4,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "scikit-learn/scikit-learn": {
          "resolved": 25,
          "total": 32
        },
        "django/django": {
          "resolved": 133,
          "total": 231
        },
        "pydata/xarray": {
          "resolved": 16,
          "total": 22
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 10
        },
        "sphinx-doc/sphinx": {
          "resolved": 23,
          "total": 44
        },
        "matplotlib/matplotlib": {
          "resolved": 21,
          "total": 34
        },
        "astropy/astropy": {
          "resolved": 10,
          "total": 22
        },
        "sympy/sympy": {
          "resolved": 48,
          "total": 75
        },
        "psf/requests": {
          "resolved": 7,
          "total": 8
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "pytest-dev/pytest": {
          "resolved": 15,
          "total": 19
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250901_entroPO_R2E_QwenCoder30BA3B_tts"
    },
    {
      "id": "verified/20250901_entroPO_R2E_QwenCoder30BA3B",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-09-01",
      "harness": "entroPO",
      "modelSlug": "R2E_QwenCoder30BA3B",
      "model": "R2E_QwenCoder30BA3B",
      "resolved": 261,
      "attempted": 273,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 52.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "matplotlib/matplotlib": {
          "resolved": 15,
          "total": 34
        },
        "psf/requests": {
          "resolved": 7,
          "total": 8
        },
        "sympy/sympy": {
          "resolved": 35,
          "total": 75
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "scikit-learn/scikit-learn": {
          "resolved": 25,
          "total": 32
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        },
        "pydata/xarray": {
          "resolved": 14,
          "total": 22
        },
        "django/django": {
          "resolved": 121,
          "total": 231
        },
        "sphinx-doc/sphinx": {
          "resolved": 20,
          "total": 44
        },
        "pytest-dev/pytest": {
          "resolved": 13,
          "total": 19
        },
        "astropy/astropy": {
          "resolved": 8,
          "total": 22
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250901_entroPO_R2E_QwenCoder30BA3B"
    },
    {
      "id": "verified/20250822_mini-v1.9.1_glm-4.5",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-08-22",
      "harness": "mini-v1.9.1",
      "modelSlug": "glm-4.5",
      "model": "GLM 4.5",
      "resolved": 271,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 54.2,
      "totalCostUsd": 148.6,
      "meanCostUsd": 0.297,
      "totalApiCalls": 20112,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250822_mini-v1.9.1_glm-4.5"
    },
    {
      "id": "verified/20250807_openhands_gpt5",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-08-07",
      "harness": "openhands",
      "modelSlug": "gpt5",
      "model": "Gpt5",
      "resolved": 359,
      "attempted": 360,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 71.8,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "astropy/astropy": {
          "resolved": 13,
          "total": 22
        },
        "sympy/sympy": {
          "resolved": 49,
          "total": 75
        },
        "psf/requests": {
          "resolved": 3,
          "total": 8
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 10
        },
        "sphinx-doc/sphinx": {
          "resolved": 28,
          "total": 44
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "matplotlib/matplotlib": {
          "resolved": 24,
          "total": 34
        },
        "pydata/xarray": {
          "resolved": 17,
          "total": 22
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "pytest-dev/pytest": {
          "resolved": 15,
          "total": 19
        },
        "scikit-learn/scikit-learn": {
          "resolved": 27,
          "total": 32
        },
        "django/django": {
          "resolved": 177,
          "total": 231
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250807_openhands_gpt5"
    },
    {
      "id": "verified/20250807_mini-v1.7.0_gpt-5",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-08-07",
      "harness": "mini-v1.7.0",
      "modelSlug": "gpt-5",
      "model": "GPT 5",
      "resolved": 325,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 65,
      "totalCostUsd": 140.19,
      "meanCostUsd": 0.28,
      "totalApiCalls": 6604,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250807_mini-v1.7.0_gpt-5"
    },
    {
      "id": "verified/20250807_mini-v1.7.0_gpt-5-mini",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-08-07",
      "harness": "mini-v1.7.0",
      "modelSlug": "gpt-5-mini",
      "model": "GPT 5 Mini",
      "resolved": 299,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 59.8,
      "totalCostUsd": 17.74,
      "meanCostUsd": 0.035,
      "totalApiCalls": 7233,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250807_mini-v1.7.0_gpt-5-mini"
    },
    {
      "id": "verified/20250807_mini-v1.7.0_kimi-k2-instruct",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-08-07",
      "harness": "mini-v1.7.0",
      "modelSlug": "kimi-k2-instruct",
      "model": "Kimi K2 Instruct",
      "resolved": 219,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 43.8,
      "totalCostUsd": 265.8,
      "meanCostUsd": 0.532,
      "totalApiCalls": 18760,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250807_mini-v1.7.0_kimi-k2-instruct"
    },
    {
      "id": "verified/20250807_mini-v1.7.0_gpt-5-nano",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-08-07",
      "harness": "mini-v1.7.0",
      "modelSlug": "gpt-5-nano",
      "model": "GPT 5 Nano",
      "resolved": 174,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 34.8,
      "totalCostUsd": 19.04,
      "meanCostUsd": 0.038,
      "totalApiCalls": 19880,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250807_mini-v1.7.0_gpt-5-nano"
    },
    {
      "id": "verified/20250807_mini-v1.7.0_gpt-oss-120b",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-08-07",
      "harness": "mini-v1.7.0",
      "modelSlug": "gpt-oss-120b",
      "model": "GPT Oss 120b",
      "resolved": 130,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 26,
      "totalCostUsd": 28.56,
      "meanCostUsd": 0.057,
      "totalApiCalls": 13799,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250807_mini-v1.7.0_gpt-oss-120b"
    },
    {
      "id": "verified/20250806_SWE-Exp_DeepSeek-V3",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-08-06",
      "harness": "SWE-Exp",
      "modelSlug": "DeepSeek-V3",
      "model": "DeepSeek V3",
      "resolved": 210,
      "attempted": 242,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 42,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 10
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "django/django": {
          "resolved": 113,
          "total": 231
        },
        "sympy/sympy": {
          "resolved": 24,
          "total": 75
        },
        "pytest-dev/pytest": {
          "resolved": 8,
          "total": 19
        },
        "astropy/astropy": {
          "resolved": 6,
          "total": 22
        },
        "psf/requests": {
          "resolved": 4,
          "total": 8
        },
        "scikit-learn/scikit-learn": {
          "resolved": 16,
          "total": 32
        },
        "sphinx-doc/sphinx": {
          "resolved": 14,
          "total": 44
        },
        "matplotlib/matplotlib": {
          "resolved": 14,
          "total": 34
        },
        "pydata/xarray": {
          "resolved": 9,
          "total": 22
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250806_SWE-Exp_DeepSeek-V3"
    },
    {
      "id": "verified/20250804_codesweep_sweagent_kimi_k2_instruct",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-08-04",
      "harness": "codesweep",
      "modelSlug": "sweagent_kimi_k2_instruct",
      "model": "Sweagent_kimi_k2_instruct",
      "resolved": 267,
      "attempted": 286,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 53.4,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sympy/sympy": {
          "resolved": 36,
          "total": 75
        },
        "sphinx-doc/sphinx": {
          "resolved": 15,
          "total": 44
        },
        "pydata/xarray": {
          "resolved": 16,
          "total": 22
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 10
        },
        "astropy/astropy": {
          "resolved": 6,
          "total": 22
        },
        "scikit-learn/scikit-learn": {
          "resolved": 24,
          "total": 32
        },
        "pytest-dev/pytest": {
          "resolved": 10,
          "total": 19
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "matplotlib/matplotlib": {
          "resolved": 15,
          "total": 34
        },
        "psf/requests": {
          "resolved": 6,
          "total": 8
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "django/django": {
          "resolved": 134,
          "total": 231
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250804_codesweep_sweagent_kimi_k2_instruct"
    },
    {
      "id": "verified/20250803_mini-v1.0.0_qwen2-5-coder-32b-instruct",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-08-03",
      "harness": "mini-v1.0.0",
      "modelSlug": "qwen2-5-coder-32b-instruct",
      "model": "Qwen2 5 Coder 32b Instruct",
      "resolved": 45,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 9,
      "totalCostUsd": 34.06,
      "meanCostUsd": 0.068,
      "totalApiCalls": 24103,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250803_mini-v1.0.0_qwen2-5-coder-32b-instruct"
    },
    {
      "id": "verified/20250802_mini-v1.0.0_claude-4-opus-20250514",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-08-02",
      "harness": "mini-v1.0.0",
      "modelSlug": "claude-4-opus-20250514",
      "model": "Claude 4 Opus 20250514",
      "resolved": 338,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 67.6,
      "totalCostUsd": 565.64,
      "meanCostUsd": 1.131,
      "totalApiCalls": 15538,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250802_mini-v1.0.0_claude-4-opus-20250514"
    },
    {
      "id": "verified/20250802_mini-v1.0.0_qwen3-coder-480b-a35b-instruct",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-08-02",
      "harness": "mini-v1.0.0",
      "modelSlug": "qwen3-coder-480b-a35b-instruct",
      "model": "Qwen3 Coder 480b A35b Instruct",
      "resolved": 277,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 55.4,
      "totalCostUsd": 123.96,
      "meanCostUsd": 0.248,
      "totalApiCalls": 32396,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250802_mini-v1.0.0_qwen3-coder-480b-a35b-instruct"
    },
    {
      "id": "verified/20250731_harness_ai",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-07-31",
      "harness": "harness",
      "modelSlug": "ai",
      "model": "Ai",
      "resolved": 374,
      "attempted": 374,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 74.8,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sympy/sympy": {
          "resolved": 54,
          "total": 75
        },
        "pydata/xarray": {
          "resolved": 18,
          "total": 22
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 10
        },
        "sphinx-doc/sphinx": {
          "resolved": 31,
          "total": 44
        },
        "pytest-dev/pytest": {
          "resolved": 16,
          "total": 19
        },
        "scikit-learn/scikit-learn": {
          "resolved": 28,
          "total": 32
        },
        "astropy/astropy": {
          "resolved": 12,
          "total": 22
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "psf/requests": {
          "resolved": 6,
          "total": 8
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "django/django": {
          "resolved": 179,
          "total": 231
        },
        "matplotlib/matplotlib": {
          "resolved": 24,
          "total": 34
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250731_harness_ai"
    },
    {
      "id": "verified/20250728_zai_glm4-5",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-07-28",
      "harness": "zai",
      "modelSlug": "glm4-5",
      "model": "Glm4 5",
      "resolved": 321,
      "attempted": 322,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 64.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "psf/requests": {
          "resolved": 1,
          "total": 8
        },
        "pydata/xarray": {
          "resolved": 17,
          "total": 22
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "sympy/sympy": {
          "resolved": 47,
          "total": 75
        },
        "scikit-learn/scikit-learn": {
          "resolved": 25,
          "total": 32
        },
        "astropy/astropy": {
          "resolved": 10,
          "total": 22
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "pytest-dev/pytest": {
          "resolved": 14,
          "total": 19
        },
        "django/django": {
          "resolved": 162,
          "total": 231
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 10
        },
        "sphinx-doc/sphinx": {
          "resolved": 23,
          "total": 44
        },
        "matplotlib/matplotlib": {
          "resolved": 16,
          "total": 34
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250728_zai_glm4-5"
    },
    {
      "id": "verified/20250726_mini-v1.0.0_claude-sonnet-4-20250514",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-07-26",
      "harness": "mini-v1.0.0",
      "modelSlug": "claude-sonnet-4-20250514",
      "model": "Claude Sonnet 4 20250514",
      "resolved": 324,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 64.8,
      "totalCostUsd": 185.73,
      "meanCostUsd": 0.371,
      "totalApiCalls": 18586,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250726_mini-v1.0.0_claude-sonnet-4-20250514"
    },
    {
      "id": "verified/20250726_mini-v1.0.0_o3-2025-04-16",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-07-26",
      "harness": "mini-v1.0.0",
      "modelSlug": "o3-2025-04-16",
      "model": "O3 2025 04 16",
      "resolved": 292,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 58.4,
      "totalCostUsd": 166.83,
      "meanCostUsd": 0.334,
      "totalApiCalls": 12349,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250726_mini-v1.0.0_o3-2025-04-16"
    },
    {
      "id": "verified/20250726_mini-v1.0.0_gemini-2.5-pro",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-07-26",
      "harness": "mini-v1.0.0",
      "modelSlug": "gemini-2.5-pro",
      "model": "Gemini 2.5 Pro",
      "resolved": 268,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 53.6,
      "totalCostUsd": 144.19,
      "meanCostUsd": 0.288,
      "totalApiCalls": 10239,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250726_mini-v1.0.0_gemini-2.5-pro"
    },
    {
      "id": "verified/20250726_mini-v1.0.0_o4-mini-2025-04-16",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-07-26",
      "harness": "mini-v1.0.0",
      "modelSlug": "o4-mini-2025-04-16",
      "model": "O4 Mini 2025 04 16",
      "resolved": 225,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 45,
      "totalCostUsd": 104.99,
      "meanCostUsd": 0.21,
      "totalApiCalls": 11639,
      "byRepo": null,
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250726_mini-v1.0.0_o4-mini-2025-04-16"
    },
    {
      "id": "verified/20250725_sweagent_devstral_small_2507",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-07-25",
      "harness": "sweagent",
      "modelSlug": "devstral_small_2507",
      "model": "Devstral_small_2507",
      "resolved": 190,
      "attempted": 196,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 38,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "astropy/astropy": {
          "resolved": 6,
          "total": 22
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 10
        },
        "matplotlib/matplotlib": {
          "resolved": 15,
          "total": 34
        },
        "sphinx-doc/sphinx": {
          "resolved": 7,
          "total": 44
        },
        "pydata/xarray": {
          "resolved": 8,
          "total": 22
        },
        "psf/requests": {
          "resolved": 2,
          "total": 8
        },
        "sympy/sympy": {
          "resolved": 28,
          "total": 75
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "scikit-learn/scikit-learn": {
          "resolved": 18,
          "total": 32
        },
        "pytest-dev/pytest": {
          "resolved": 7,
          "total": 19
        },
        "django/django": {
          "resolved": 94,
          "total": 231
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250725_sweagent_devstral_small_2507"
    },
    {
      "id": "verified/20250720_Lingxi-v1.5_claude-4-sonnet-20250514",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-07-20",
      "harness": "Lingxi-v1.5",
      "modelSlug": "claude-4-sonnet-20250514",
      "model": "Claude 4 Sonnet 20250514",
      "resolved": 373,
      "attempted": 374,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 74.6,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sphinx-doc/sphinx": {
          "resolved": 29,
          "total": 44
        },
        "astropy/astropy": {
          "resolved": 12,
          "total": 22
        },
        "pydata/xarray": {
          "resolved": 18,
          "total": 22
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        },
        "psf/requests": {
          "resolved": 6,
          "total": 8
        },
        "matplotlib/matplotlib": {
          "resolved": 24,
          "total": 34
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "django/django": {
          "resolved": 178,
          "total": 231
        },
        "sympy/sympy": {
          "resolved": 57,
          "total": 75
        },
        "scikit-learn/scikit-learn": {
          "resolved": 28,
          "total": 32
        },
        "pytest-dev/pytest": {
          "resolved": 17,
          "total": 19
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250720_Lingxi-v1.5_claude-4-sonnet-20250514"
    },
    {
      "id": "verified/20250716_openhands_kimi_k2",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-07-16",
      "harness": "openhands",
      "modelSlug": "kimi_k2",
      "model": "Kimi_k2",
      "resolved": 327,
      "attempted": 328,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 65.4,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "astropy/astropy": {
          "resolved": 11,
          "total": 22
        },
        "sympy/sympy": {
          "resolved": 53,
          "total": 75
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "pydata/xarray": {
          "resolved": 15,
          "total": 22
        },
        "sphinx-doc/sphinx": {
          "resolved": 26,
          "total": 44
        },
        "psf/requests": {
          "resolved": 2,
          "total": 8
        },
        "scikit-learn/scikit-learn": {
          "resolved": 26,
          "total": 32
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 10
        },
        "matplotlib/matplotlib": {
          "resolved": 20,
          "total": 34
        },
        "pytest-dev/pytest": {
          "resolved": 12,
          "total": 19
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "django/django": {
          "resolved": 157,
          "total": 231
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250716_openhands_kimi_k2"
    },
    {
      "id": "verified/20250715_qodo_command",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-07-15",
      "harness": "qodo",
      "modelSlug": "command",
      "model": "Command",
      "resolved": 356,
      "attempted": 356,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 71.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "scikit-learn/scikit-learn": {
          "resolved": 28,
          "total": 32
        },
        "django/django": {
          "resolved": 172,
          "total": 231
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "pydata/xarray": {
          "resolved": 15,
          "total": 22
        },
        "psf/requests": {
          "resolved": 3,
          "total": 8
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 10
        },
        "astropy/astropy": {
          "resolved": 12,
          "total": 22
        },
        "sphinx-doc/sphinx": {
          "resolved": 31,
          "total": 44
        },
        "matplotlib/matplotlib": {
          "resolved": 23,
          "total": 34
        },
        "sympy/sympy": {
          "resolved": 50,
          "total": 75
        },
        "pytest-dev/pytest": {
          "resolved": 16,
          "total": 19
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250715_qodo_command"
    },
    {
      "id": "verified/20250629_deepswerl_r2eagent_tts",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-06-29",
      "harness": "deepswerl",
      "modelSlug": "r2eagent_tts",
      "model": "R2eagent_tts",
      "resolved": 294,
      "attempted": 297,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 58.8,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "psf/requests": {
          "resolved": 4,
          "total": 8
        },
        "django/django": {
          "resolved": 134,
          "total": 231
        },
        "sphinx-doc/sphinx": {
          "resolved": 22,
          "total": 44
        },
        "astropy/astropy": {
          "resolved": 11,
          "total": 22
        },
        "pydata/xarray": {
          "resolved": 17,
          "total": 22
        },
        "pytest-dev/pytest": {
          "resolved": 13,
          "total": 19
        },
        "sympy/sympy": {
          "resolved": 45,
          "total": 75
        },
        "scikit-learn/scikit-learn": {
          "resolved": 24,
          "total": 32
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        },
        "matplotlib/matplotlib": {
          "resolved": 21,
          "total": 34
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250629_deepswerl_r2eagent_tts"
    },
    {
      "id": "verified/20250629_deepswerl_r2eagent",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-06-29",
      "harness": "deepswerl",
      "modelSlug": "r2eagent",
      "model": "R2eagent",
      "resolved": 211,
      "attempted": 256,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 42.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pytest-dev/pytest": {
          "resolved": 11,
          "total": 19
        },
        "scikit-learn/scikit-learn": {
          "resolved": 21,
          "total": 32
        },
        "astropy/astropy": {
          "resolved": 5,
          "total": 22
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "pydata/xarray": {
          "resolved": 11,
          "total": 22
        },
        "sympy/sympy": {
          "resolved": 28,
          "total": 75
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "sphinx-doc/sphinx": {
          "resolved": 15,
          "total": 44
        },
        "matplotlib/matplotlib": {
          "resolved": 11,
          "total": 34
        },
        "psf/requests": {
          "resolved": 3,
          "total": 8
        },
        "django/django": {
          "resolved": 103,
          "total": 231
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250629_deepswerl_r2eagent"
    },
    {
      "id": "verified/20250627_agentless_MCTS-Refine-7B",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-06-27",
      "harness": "agentless",
      "modelSlug": "MCTS-Refine-7B",
      "model": "MCTS Refine 7B",
      "resolved": 116,
      "attempted": 155,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 23.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "scikit-learn/scikit-learn": {
          "resolved": 17,
          "total": 32
        },
        "sympy/sympy": {
          "resolved": 14,
          "total": 75
        },
        "matplotlib/matplotlib": {
          "resolved": 8,
          "total": 34
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "sphinx-doc/sphinx": {
          "resolved": 11,
          "total": 44
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "psf/requests": {
          "resolved": 2,
          "total": 8
        },
        "pytest-dev/pytest": {
          "resolved": 7,
          "total": 19
        },
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 10
        },
        "django/django": {
          "resolved": 37,
          "total": 231
        },
        "astropy/astropy": {
          "resolved": 9,
          "total": 22
        },
        "pydata/xarray": {
          "resolved": 9,
          "total": 22
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250627_agentless_MCTS-Refine-7B"
    },
    {
      "id": "verified/20250616_Skywork-SWE-32B+TTS_Bo8",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-06-16",
      "harness": "Skywork-SWE-32B+TTS",
      "modelSlug": "Bo8",
      "model": "Bo8",
      "resolved": 235,
      "attempted": 236,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 47,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pydata/xarray": {
          "resolved": 14,
          "total": 22
        },
        "matplotlib/matplotlib": {
          "resolved": 15,
          "total": 34
        },
        "scikit-learn/scikit-learn": {
          "resolved": 22,
          "total": 32
        },
        "pytest-dev/pytest": {
          "resolved": 10,
          "total": 19
        },
        "astropy/astropy": {
          "resolved": 8,
          "total": 22
        },
        "sympy/sympy": {
          "resolved": 31,
          "total": 75
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "django/django": {
          "resolved": 115,
          "total": 231
        },
        "sphinx-doc/sphinx": {
          "resolved": 12,
          "total": 44
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 10
        },
        "psf/requests": {
          "resolved": 3,
          "total": 8
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250616_Skywork-SWE-32B+TTS_Bo8"
    },
    {
      "id": "verified/20250611_moatless_claude-4-sonnet-20250514",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-06-11",
      "harness": "moatless",
      "modelSlug": "claude-4-sonnet-20250514",
      "model": "Claude 4 Sonnet 20250514",
      "resolved": 354,
      "attempted": 357,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 70.8,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "django/django": {
          "resolved": 172,
          "total": 231
        },
        "pylint-dev/pylint": {
          "resolved": 6,
          "total": 10
        },
        "scikit-learn/scikit-learn": {
          "resolved": 26,
          "total": 32
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "sphinx-doc/sphinx": {
          "resolved": 27,
          "total": 44
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "psf/requests": {
          "resolved": 5,
          "total": 8
        },
        "sympy/sympy": {
          "resolved": 51,
          "total": 75
        },
        "astropy/astropy": {
          "resolved": 11,
          "total": 22
        },
        "pydata/xarray": {
          "resolved": 16,
          "total": 22
        },
        "matplotlib/matplotlib": {
          "resolved": 23,
          "total": 34
        },
        "pytest-dev/pytest": {
          "resolved": 15,
          "total": 19
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250611_moatless_claude-4-sonnet-20250514"
    },
    {
      "id": "verified/20250610_augment_agent_v1",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-06-10",
      "harness": "augment",
      "modelSlug": "agent_v1",
      "model": "Agent_v1",
      "resolved": 352,
      "attempted": 352,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 70.4,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "astropy/astropy": {
          "resolved": 13,
          "total": 22
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "sphinx-doc/sphinx": {
          "resolved": 26,
          "total": 44
        },
        "pydata/xarray": {
          "resolved": 10,
          "total": 22
        },
        "pytest-dev/pytest": {
          "resolved": 15,
          "total": 19
        },
        "psf/requests": {
          "resolved": 5,
          "total": 8
        },
        "django/django": {
          "resolved": 170,
          "total": 231
        },
        "matplotlib/matplotlib": {
          "resolved": 23,
          "total": 34
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 10
        },
        "sympy/sympy": {
          "resolved": 55,
          "total": 75
        },
        "scikit-learn/scikit-learn": {
          "resolved": 29,
          "total": 32
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250610_augment_agent_v1"
    },
    {
      "id": "verified/20250603_Refact_Agent_claude-4-sonnet",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-06-03",
      "harness": "Refact",
      "modelSlug": "Agent_claude-4-sonnet",
      "model": "Agent_claude 4 Sonnet",
      "resolved": 372,
      "attempted": 373,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 74.4,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "psf/requests": {
          "resolved": 7,
          "total": 8
        },
        "sympy/sympy": {
          "resolved": 55,
          "total": 75
        },
        "astropy/astropy": {
          "resolved": 12,
          "total": 22
        },
        "pytest-dev/pytest": {
          "resolved": 15,
          "total": 19
        },
        "scikit-learn/scikit-learn": {
          "resolved": 27,
          "total": 32
        },
        "sphinx-doc/sphinx": {
          "resolved": 31,
          "total": 44
        },
        "pylint-dev/pylint": {
          "resolved": 5,
          "total": 10
        },
        "matplotlib/matplotlib": {
          "resolved": 23,
          "total": 34
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 2
        },
        "pydata/xarray": {
          "resolved": 18,
          "total": 22
        },
        "django/django": {
          "resolved": 176,
          "total": 231
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250603_Refact_Agent_claude-4-sonnet"
    },
    {
      "id": "verified/20250528_patchpilot_Co-PatcheR",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-05-28",
      "harness": "patchpilot",
      "modelSlug": "Co-PatcheR",
      "model": "Co PatcheR",
      "resolved": 230,
      "attempted": 230,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 46,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "scikit-learn/scikit-learn": {
          "resolved": 24,
          "total": 32
        },
        "sympy/sympy": {
          "resolved": 34,
          "total": 75
        },
        "astropy/astropy": {
          "resolved": 7,
          "total": 22
        },
        "pydata/xarray": {
          "resolved": 8,
          "total": 22
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        },
        "pytest-dev/pytest": {
          "resolved": 9,
          "total": 19
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "matplotlib/matplotlib": {
          "resolved": 13,
          "total": 34
        },
        "django/django": {
          "resolved": 115,
          "total": 231
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "psf/requests": {
          "resolved": 3,
          "total": 8
        },
        "sphinx-doc/sphinx": {
          "resolved": 14,
          "total": 44
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250528_patchpilot_Co-PatcheR"
    },
    {
      "id": "verified/20250524_openhands_claude_4_sonnet",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-05-24",
      "harness": "openhands",
      "modelSlug": "claude_4_sonnet",
      "model": "Claude_4_sonnet",
      "resolved": 352,
      "attempted": 353,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 70.4,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "matplotlib/matplotlib": {
          "resolved": 20,
          "total": 34
        },
        "pytest-dev/pytest": {
          "resolved": 13,
          "total": 19
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 2
        },
        "scikit-learn/scikit-learn": {
          "resolved": 28,
          "total": 32
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 10
        },
        "psf/requests": {
          "resolved": 1,
          "total": 8
        },
        "sympy/sympy": {
          "resolved": 54,
          "total": 75
        },
        "pydata/xarray": {
          "resolved": 18,
          "total": 22
        },
        "astropy/astropy": {
          "resolved": 12,
          "total": 22
        },
        "django/django": {
          "resolved": 171,
          "total": 231
        },
        "sphinx-doc/sphinx": {
          "resolved": 28,
          "total": 44
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250524_openhands_claude_4_sonnet"
    },
    {
      "id": "verified/20250522_tools_claude-4-opus",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-05-22",
      "harness": "tools",
      "modelSlug": "claude-4-opus",
      "model": "Claude 4 Opus",
      "resolved": 366,
      "attempted": 366,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 73.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "django/django": {
          "resolved": 181,
          "total": 231
        },
        "psf/requests": {
          "resolved": 6,
          "total": 8
        },
        "astropy/astropy": {
          "resolved": 13,
          "total": 22
        },
        "pydata/xarray": {
          "resolved": 18,
          "total": 22
        },
        "sphinx-doc/sphinx": {
          "resolved": 28,
          "total": 44
        },
        "matplotlib/matplotlib": {
          "resolved": 21,
          "total": 34
        },
        "pytest-dev/pytest": {
          "resolved": 16,
          "total": 19
        },
        "scikit-learn/scikit-learn": {
          "resolved": 28,
          "total": 32
        },
        "sympy/sympy": {
          "resolved": 51,
          "total": 75
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250522_tools_claude-4-opus"
    },
    {
      "id": "verified/20250522_tools_claude-4-sonnet",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-05-22",
      "harness": "tools",
      "modelSlug": "claude-4-sonnet",
      "model": "Claude 4 Sonnet",
      "resolved": 362,
      "attempted": 362,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 72.4,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pytest-dev/pytest": {
          "resolved": 17,
          "total": 19
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        },
        "scikit-learn/scikit-learn": {
          "resolved": 28,
          "total": 32
        },
        "pydata/xarray": {
          "resolved": 18,
          "total": 22
        },
        "matplotlib/matplotlib": {
          "resolved": 22,
          "total": 34
        },
        "sympy/sympy": {
          "resolved": 53,
          "total": 75
        },
        "django/django": {
          "resolved": 177,
          "total": 231
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "psf/requests": {
          "resolved": 4,
          "total": 8
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "astropy/astropy": {
          "resolved": 11,
          "total": 22
        },
        "sphinx-doc/sphinx": {
          "resolved": 28,
          "total": 44
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250522_tools_claude-4-sonnet"
    },
    {
      "id": "verified/20250522_sweagent_claude-4-sonnet-20250514",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-05-22",
      "harness": "sweagent",
      "modelSlug": "claude-4-sonnet-20250514",
      "model": "Claude 4 Sonnet 20250514",
      "resolved": 333,
      "attempted": 333,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 66.6,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 10
        },
        "sphinx-doc/sphinx": {
          "resolved": 28,
          "total": 44
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "pytest-dev/pytest": {
          "resolved": 13,
          "total": 19
        },
        "django/django": {
          "resolved": 168,
          "total": 231
        },
        "scikit-learn/scikit-learn": {
          "resolved": 25,
          "total": 32
        },
        "sympy/sympy": {
          "resolved": 47,
          "total": 75
        },
        "pydata/xarray": {
          "resolved": 17,
          "total": 22
        },
        "matplotlib/matplotlib": {
          "resolved": 15,
          "total": 34
        },
        "astropy/astropy": {
          "resolved": 11,
          "total": 22
        },
        "mwaskom/seaborn": {
          "resolved": 2,
          "total": 2
        },
        "psf/requests": {
          "resolved": 5,
          "total": 8
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250522_sweagent_claude-4-sonnet-20250514"
    },
    {
      "id": "verified/20250520_openhands_devstral_small",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-05-20",
      "harness": "openhands",
      "modelSlug": "devstral_small",
      "model": "Devstral_small",
      "resolved": 234,
      "attempted": 242,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 46.8,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pytest-dev/pytest": {
          "resolved": 10,
          "total": 19
        },
        "pydata/xarray": {
          "resolved": 11,
          "total": 22
        },
        "django/django": {
          "resolved": 117,
          "total": 231
        },
        "psf/requests": {
          "resolved": 0,
          "total": 8
        },
        "sympy/sympy": {
          "resolved": 31,
          "total": 75
        },
        "astropy/astropy": {
          "resolved": 8,
          "total": 22
        },
        "sphinx-doc/sphinx": {
          "resolved": 16,
          "total": 44
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "scikit-learn/scikit-learn": {
          "resolved": 23,
          "total": 32
        },
        "matplotlib/matplotlib": {
          "resolved": 15,
          "total": 34
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250520_openhands_devstral_small"
    },
    {
      "id": "verified/20250516_cortexa_o3",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-05-16",
      "harness": "cortexa",
      "modelSlug": "o3",
      "model": "O3",
      "resolved": 341,
      "attempted": 462,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 68.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 10
        },
        "sphinx-doc/sphinx": {
          "resolved": 29,
          "total": 44
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "pytest-dev/pytest": {
          "resolved": 15,
          "total": 19
        },
        "django/django": {
          "resolved": 168,
          "total": 231
        },
        "scikit-learn/scikit-learn": {
          "resolved": 27,
          "total": 32
        },
        "sympy/sympy": {
          "resolved": 47,
          "total": 75
        },
        "matplotlib/matplotlib": {
          "resolved": 20,
          "total": 34
        },
        "psf/requests": {
          "resolved": 3,
          "total": 8
        },
        "pydata/xarray": {
          "resolved": 16,
          "total": 22
        },
        "astropy/astropy": {
          "resolved": 12,
          "total": 22
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250516_cortexa_o3"
    },
    {
      "id": "verified/20250515_Refact_Agent",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-05-15",
      "harness": "Refact",
      "modelSlug": "Agent",
      "model": "Agent",
      "resolved": 352,
      "attempted": 353,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 70.4,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "astropy/astropy": {
          "resolved": 12,
          "total": 22
        },
        "pydata/xarray": {
          "resolved": 18,
          "total": 22
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "sympy/sympy": {
          "resolved": 54,
          "total": 75
        },
        "sphinx-doc/sphinx": {
          "resolved": 28,
          "total": 44
        },
        "matplotlib/matplotlib": {
          "resolved": 20,
          "total": 34
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "psf/requests": {
          "resolved": 6,
          "total": 8
        },
        "django/django": {
          "resolved": 165,
          "total": 231
        },
        "scikit-learn/scikit-learn": {
          "resolved": 28,
          "total": 32
        },
        "pytest-dev/pytest": {
          "resolved": 16,
          "total": 19
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 10
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250515_Refact_Agent"
    },
    {
      "id": "verified/20250514_aime_coder",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-05-14",
      "harness": "aime",
      "modelSlug": "coder",
      "model": "Coder",
      "resolved": 332,
      "attempted": 332,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 66.4,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "sympy/sympy": {
          "resolved": 51,
          "total": 75
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "astropy/astropy": {
          "resolved": 12,
          "total": 22
        },
        "pydata/xarray": {
          "resolved": 18,
          "total": 22
        },
        "pytest-dev/pytest": {
          "resolved": 13,
          "total": 19
        },
        "matplotlib/matplotlib": {
          "resolved": 17,
          "total": 34
        },
        "psf/requests": {
          "resolved": 3,
          "total": 8
        },
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 10
        },
        "sphinx-doc/sphinx": {
          "resolved": 23,
          "total": 44
        },
        "django/django": {
          "resolved": 166,
          "total": 231
        },
        "scikit-learn/scikit-learn": {
          "resolved": 27,
          "total": 32
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250514_aime_coder"
    },
    {
      "id": "verified/20250511_sweagent_lm_32b",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-05-11",
      "harness": "sweagent",
      "modelSlug": "lm_32b",
      "model": "Lm_32b",
      "resolved": 201,
      "attempted": 205,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 40.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "sympy/sympy": {
          "resolved": 23,
          "total": 75
        },
        "sphinx-doc/sphinx": {
          "resolved": 12,
          "total": 44
        },
        "scikit-learn/scikit-learn": {
          "resolved": 19,
          "total": 32
        },
        "django/django": {
          "resolved": 95,
          "total": 231
        },
        "psf/requests": {
          "resolved": 4,
          "total": 8
        },
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 10
        },
        "pydata/xarray": {
          "resolved": 11,
          "total": 22
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "matplotlib/matplotlib": {
          "resolved": 15,
          "total": 34
        },
        "pytest-dev/pytest": {
          "resolved": 11,
          "total": 19
        },
        "astropy/astropy": {
          "resolved": 9,
          "total": 22
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250511_sweagent_lm_32b"
    },
    {
      "id": "verified/20250430_zencoder_ai",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-04-30",
      "harness": "zencoder",
      "modelSlug": "ai",
      "model": "Ai",
      "resolved": 350,
      "attempted": 351,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 70,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pylint-dev/pylint": {
          "resolved": 5,
          "total": 10
        },
        "pydata/xarray": {
          "resolved": 16,
          "total": 22
        },
        "pytest-dev/pytest": {
          "resolved": 14,
          "total": 19
        },
        "scikit-learn/scikit-learn": {
          "resolved": 26,
          "total": 32
        },
        "psf/requests": {
          "resolved": 4,
          "total": 8
        },
        "sympy/sympy": {
          "resolved": 49,
          "total": 75
        },
        "matplotlib/matplotlib": {
          "resolved": 20,
          "total": 34
        },
        "astropy/astropy": {
          "resolved": 11,
          "total": 22
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "sphinx-doc/sphinx": {
          "resolved": 29,
          "total": 44
        },
        "django/django": {
          "resolved": 174,
          "total": 231
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250430_zencoder_ai"
    },
    {
      "id": "verified/20250405_swe-rizzo_claude37",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-04-05",
      "harness": "swe-rizzo",
      "modelSlug": "claude37",
      "model": "Claude37",
      "resolved": 283,
      "attempted": 283,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 56.6,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "django/django": {
          "resolved": 144,
          "total": 231
        },
        "sphinx-doc/sphinx": {
          "resolved": 19,
          "total": 44
        },
        "sympy/sympy": {
          "resolved": 42,
          "total": 75
        },
        "pydata/xarray": {
          "resolved": 9,
          "total": 22
        },
        "pytest-dev/pytest": {
          "resolved": 12,
          "total": 19
        },
        "scikit-learn/scikit-learn": {
          "resolved": 24,
          "total": 32
        },
        "astropy/astropy": {
          "resolved": 11,
          "total": 22
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        },
        "psf/requests": {
          "resolved": 3,
          "total": 8
        },
        "matplotlib/matplotlib": {
          "resolved": 15,
          "total": 34
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250405_swe-rizzo_claude37"
    },
    {
      "id": "verified/20250316_augment_agent_v0",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-03-16",
      "harness": "augment",
      "modelSlug": "agent_v0",
      "model": "Agent_v0",
      "resolved": 327,
      "attempted": 327,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 65.4,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "astropy/astropy": {
          "resolved": 12,
          "total": 22
        },
        "matplotlib/matplotlib": {
          "resolved": 19,
          "total": 34
        },
        "django/django": {
          "resolved": 156,
          "total": 231
        },
        "scikit-learn/scikit-learn": {
          "resolved": 28,
          "total": 32
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 10
        },
        "pydata/xarray": {
          "resolved": 12,
          "total": 22
        },
        "sphinx-doc/sphinx": {
          "resolved": 26,
          "total": 44
        },
        "pytest-dev/pytest": {
          "resolved": 13,
          "total": 19
        },
        "psf/requests": {
          "resolved": 6,
          "total": 8
        },
        "sympy/sympy": {
          "resolved": 50,
          "total": 75
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250316_augment_agent_v0"
    },
    {
      "id": "verified/20250306_SWE-Fixer_Qwen2.5-7b-retriever_Qwen2.5-72b-editor",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-03-06",
      "harness": "SWE-Fixer",
      "modelSlug": "Qwen2.5-7b-retriever_Qwen2.5-72b-editor",
      "model": "Qwen2.5 7b Retriever_Qwen2.5 72b Editor",
      "resolved": 164,
      "attempted": 183,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 32.8,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "astropy/astropy": {
          "resolved": 5,
          "total": 22
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        },
        "pytest-dev/pytest": {
          "resolved": 4,
          "total": 19
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "pydata/xarray": {
          "resolved": 7,
          "total": 22
        },
        "django/django": {
          "resolved": 87,
          "total": 231
        },
        "matplotlib/matplotlib": {
          "resolved": 8,
          "total": 34
        },
        "sphinx-doc/sphinx": {
          "resolved": 12,
          "total": 44
        },
        "sympy/sympy": {
          "resolved": 16,
          "total": 75
        },
        "psf/requests": {
          "resolved": 4,
          "total": 8
        },
        "scikit-learn/scikit-learn": {
          "resolved": 18,
          "total": 32
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250306_SWE-Fixer_Qwen2.5-7b-retriever_Qwen2.5-72b-editor"
    },
    {
      "id": "verified/20250226_swerl_llama3_70b",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-02-26",
      "harness": "swerl",
      "modelSlug": "llama3_70b",
      "model": "Llama3_70b",
      "resolved": 206,
      "attempted": 207,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 41.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "matplotlib/matplotlib": {
          "resolved": 17,
          "total": 34
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 10
        },
        "pytest-dev/pytest": {
          "resolved": 9,
          "total": 19
        },
        "pydata/xarray": {
          "resolved": 14,
          "total": 22
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "django/django": {
          "resolved": 91,
          "total": 231
        },
        "psf/requests": {
          "resolved": 5,
          "total": 8
        },
        "scikit-learn/scikit-learn": {
          "resolved": 23,
          "total": 32
        },
        "sphinx-doc/sphinx": {
          "resolved": 8,
          "total": 44
        },
        "astropy/astropy": {
          "resolved": 7,
          "total": 22
        },
        "sympy/sympy": {
          "resolved": 28,
          "total": 75
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250226_swerl_llama3_70b"
    },
    {
      "id": "verified/20250225_sweagent_claude-3-7-sonnet",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-02-25",
      "harness": "sweagent",
      "modelSlug": "claude-3-7-sonnet",
      "model": "Claude 3.7 Sonnet",
      "resolved": 312,
      "attempted": 314,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 62.4,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sphinx-doc/sphinx": {
          "resolved": 20,
          "total": 44
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 10
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "pytest-dev/pytest": {
          "resolved": 11,
          "total": 19
        },
        "psf/requests": {
          "resolved": 7,
          "total": 8
        },
        "astropy/astropy": {
          "resolved": 12,
          "total": 22
        },
        "matplotlib/matplotlib": {
          "resolved": 19,
          "total": 34
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "django/django": {
          "resolved": 149,
          "total": 231
        },
        "pydata/xarray": {
          "resolved": 16,
          "total": 22
        },
        "sympy/sympy": {
          "resolved": 47,
          "total": 75
        },
        "scikit-learn/scikit-learn": {
          "resolved": 27,
          "total": 32
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250225_sweagent_claude-3-7-sonnet"
    },
    {
      "id": "verified/20250224_tools_claude-3-7-sonnet",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-02-24",
      "harness": "tools",
      "modelSlug": "claude-3-7-sonnet",
      "model": "Claude 3.7 Sonnet",
      "resolved": 316,
      "attempted": 316,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 63.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sphinx-doc/sphinx": {
          "resolved": 25,
          "total": 44
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 10
        },
        "scikit-learn/scikit-learn": {
          "resolved": 27,
          "total": 32
        },
        "pytest-dev/pytest": {
          "resolved": 13,
          "total": 19
        },
        "matplotlib/matplotlib": {
          "resolved": 16,
          "total": 34
        },
        "pydata/xarray": {
          "resolved": 17,
          "total": 22
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "astropy/astropy": {
          "resolved": 11,
          "total": 22
        },
        "psf/requests": {
          "resolved": 4,
          "total": 8
        },
        "django/django": {
          "resolved": 154,
          "total": 231
        },
        "sympy/sympy": {
          "resolved": 44,
          "total": 75
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250224_tools_claude-3-7-sonnet"
    },
    {
      "id": "verified/20250214_agentless_lite_o3_mini",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-02-14",
      "harness": "agentless",
      "modelSlug": "lite_o3_mini",
      "model": "Lite_o3_mini",
      "resolved": 212,
      "attempted": 222,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 42.4,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "django/django": {
          "resolved": 119,
          "total": 231
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "astropy/astropy": {
          "resolved": 4,
          "total": 22
        },
        "pytest-dev/pytest": {
          "resolved": 8,
          "total": 19
        },
        "sympy/sympy": {
          "resolved": 28,
          "total": 75
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 10
        },
        "pydata/xarray": {
          "resolved": 8,
          "total": 22
        },
        "matplotlib/matplotlib": {
          "resolved": 5,
          "total": 34
        },
        "sphinx-doc/sphinx": {
          "resolved": 17,
          "total": 44
        },
        "psf/requests": {
          "resolved": 4,
          "total": 8
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "scikit-learn/scikit-learn": {
          "resolved": 15,
          "total": 32
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250214_agentless_lite_o3_mini"
    },
    {
      "id": "verified/20250203_openhands_4x_scaled",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-02-03",
      "harness": "openhands",
      "modelSlug": "4x_scaled",
      "model": "4x_scaled",
      "resolved": 304,
      "attempted": 306,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 60.8,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sphinx-doc/sphinx": {
          "resolved": 22,
          "total": 44
        },
        "matplotlib/matplotlib": {
          "resolved": 17,
          "total": 34
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "psf/requests": {
          "resolved": 5,
          "total": 8
        },
        "scikit-learn/scikit-learn": {
          "resolved": 26,
          "total": 32
        },
        "django/django": {
          "resolved": 146,
          "total": 231
        },
        "sympy/sympy": {
          "resolved": 47,
          "total": 75
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 10
        },
        "astropy/astropy": {
          "resolved": 11,
          "total": 22
        },
        "pydata/xarray": {
          "resolved": 14,
          "total": 22
        },
        "pytest-dev/pytest": {
          "resolved": 12,
          "total": 19
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250203_openhands_4x_scaled"
    },
    {
      "id": "verified/20250118_codeshellagent_gemini_2.0_flash_experimental",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-01-18",
      "harness": "codeshellagent",
      "modelSlug": "gemini_2.0_flash_experimental",
      "model": "Gemini_2.0_flash_experimental",
      "resolved": 221,
      "attempted": 235,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 44.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "astropy/astropy": {
          "resolved": 9,
          "total": 22
        },
        "psf/requests": {
          "resolved": 3,
          "total": 8
        },
        "django/django": {
          "resolved": 107,
          "total": 231
        },
        "scikit-learn/scikit-learn": {
          "resolved": 19,
          "total": 32
        },
        "sympy/sympy": {
          "resolved": 30,
          "total": 75
        },
        "pydata/xarray": {
          "resolved": 10,
          "total": 22
        },
        "sphinx-doc/sphinx": {
          "resolved": 13,
          "total": 44
        },
        "pytest-dev/pytest": {
          "resolved": 8,
          "total": 19
        },
        "matplotlib/matplotlib": {
          "resolved": 19,
          "total": 34
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250118_codeshellagent_gemini_2.0_flash_experimental"
    },
    {
      "id": "verified/20250117_wandb_programmer_o1_crosscheck5",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-01-17",
      "harness": "wandb",
      "modelSlug": "programmer_o1_crosscheck5",
      "model": "Programmer_o1_crosscheck5",
      "resolved": 323,
      "attempted": 324,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 64.6,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "astropy/astropy": {
          "resolved": 12,
          "total": 22
        },
        "matplotlib/matplotlib": {
          "resolved": 19,
          "total": 34
        },
        "sympy/sympy": {
          "resolved": 47,
          "total": 75
        },
        "django/django": {
          "resolved": 154,
          "total": 231
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "scikit-learn/scikit-learn": {
          "resolved": 27,
          "total": 32
        },
        "psf/requests": {
          "resolved": 5,
          "total": 8
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 10
        },
        "pytest-dev/pytest": {
          "resolved": 14,
          "total": 19
        },
        "pydata/xarray": {
          "resolved": 18,
          "total": 22
        },
        "sphinx-doc/sphinx": {
          "resolved": 23,
          "total": 44
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250117_wandb_programmer_o1_crosscheck5"
    },
    {
      "id": "verified/20250110_blackboxai_agent_v1.1",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-01-10",
      "harness": "blackboxai",
      "modelSlug": "agent_v1.1",
      "model": "Agent_v1.1",
      "resolved": 314,
      "attempted": 334,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 62.8,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 10
        },
        "pytest-dev/pytest": {
          "resolved": 13,
          "total": 19
        },
        "pydata/xarray": {
          "resolved": 13,
          "total": 22
        },
        "django/django": {
          "resolved": 155,
          "total": 231
        },
        "psf/requests": {
          "resolved": 4,
          "total": 8
        },
        "scikit-learn/scikit-learn": {
          "resolved": 24,
          "total": 32
        },
        "sympy/sympy": {
          "resolved": 44,
          "total": 75
        },
        "matplotlib/matplotlib": {
          "resolved": 21,
          "total": 34
        },
        "astropy/astropy": {
          "resolved": 11,
          "total": 22
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "sphinx-doc/sphinx": {
          "resolved": 25,
          "total": 44
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250110_blackboxai_agent_v1.1"
    },
    {
      "id": "verified/20250110_learn_by_interact_claude3.5",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2025-01-10",
      "harness": "learn",
      "modelSlug": "by_interact_claude3.5",
      "model": "By_interact_claude3.5",
      "resolved": 301,
      "attempted": 391,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 60.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sympy/sympy": {
          "resolved": 48,
          "total": 75
        },
        "psf/requests": {
          "resolved": 6,
          "total": 8
        },
        "sphinx-doc/sphinx": {
          "resolved": 18,
          "total": 44
        },
        "matplotlib/matplotlib": {
          "resolved": 15,
          "total": 34
        },
        "mwaskom/seaborn": {
          "resolved": 1,
          "total": 2
        },
        "pydata/xarray": {
          "resolved": 15,
          "total": 22
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 10
        },
        "pytest-dev/pytest": {
          "resolved": 12,
          "total": 19
        },
        "django/django": {
          "resolved": 147,
          "total": 231
        },
        "scikit-learn/scikit-learn": {
          "resolved": 26,
          "total": 32
        },
        "astropy/astropy": {
          "resolved": 8,
          "total": 22
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20250110_learn_by_interact_claude3.5"
    },
    {
      "id": "verified/20241221_codestory_midwit_claude-3-5-sonnet_swe-search",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-12-21",
      "harness": "codestory",
      "modelSlug": "midwit_claude-3-5-sonnet_swe-search",
      "model": "Midwit_claude 3.5 Sonnet_swe Search",
      "resolved": 311,
      "attempted": 356,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 62.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "psf/requests": {
          "resolved": 6,
          "total": 8
        },
        "matplotlib/matplotlib": {
          "resolved": 21,
          "total": 34
        },
        "scikit-learn/scikit-learn": {
          "resolved": 25,
          "total": 32
        },
        "sympy/sympy": {
          "resolved": 42,
          "total": 75
        },
        "pydata/xarray": {
          "resolved": 13,
          "total": 22
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 10
        },
        "astropy/astropy": {
          "resolved": 10,
          "total": 22
        },
        "sphinx-doc/sphinx": {
          "resolved": 19,
          "total": 44
        },
        "django/django": {
          "resolved": 159,
          "total": 231
        },
        "pytest-dev/pytest": {
          "resolved": 12,
          "total": 19
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20241221_codestory_midwit_claude-3-5-sonnet_swe-search"
    },
    {
      "id": "verified/20241212_google_jules_gemini_2.0_flash_experimental",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-12-12",
      "harness": "google",
      "modelSlug": "jules_gemini_2.0_flash_experimental",
      "model": "Jules_gemini_2.0_flash_experimental",
      "resolved": 261,
      "attempted": 261,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 52.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "astropy/astropy": {
          "resolved": 12,
          "total": 22
        },
        "psf/requests": {
          "resolved": 4,
          "total": 8
        },
        "sympy/sympy": {
          "resolved": 37,
          "total": 75
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        },
        "matplotlib/matplotlib": {
          "resolved": 18,
          "total": 34
        },
        "sphinx-doc/sphinx": {
          "resolved": 16,
          "total": 44
        },
        "pydata/xarray": {
          "resolved": 12,
          "total": 22
        },
        "django/django": {
          "resolved": 121,
          "total": 231
        },
        "pytest-dev/pytest": {
          "resolved": 13,
          "total": 19
        },
        "scikit-learn/scikit-learn": {
          "resolved": 25,
          "total": 32
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20241212_google_jules_gemini_2.0_flash_experimental"
    },
    {
      "id": "verified/20241202_agentless-1.5_claude-3.5-sonnet-20241022",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-12-02",
      "harness": "agentless-1.5",
      "modelSlug": "claude-3.5-sonnet-20241022",
      "model": "Claude 3.5 Sonnet 20241022",
      "resolved": 254,
      "attempted": 257,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 50.8,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pydata/xarray": {
          "resolved": 13,
          "total": 22
        },
        "astropy/astropy": {
          "resolved": 8,
          "total": 22
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        },
        "sympy/sympy": {
          "resolved": 33,
          "total": 75
        },
        "pytest-dev/pytest": {
          "resolved": 12,
          "total": 19
        },
        "psf/requests": {
          "resolved": 6,
          "total": 8
        },
        "scikit-learn/scikit-learn": {
          "resolved": 25,
          "total": 32
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "sphinx-doc/sphinx": {
          "resolved": 16,
          "total": 44
        },
        "matplotlib/matplotlib": {
          "resolved": 17,
          "total": 34
        },
        "django/django": {
          "resolved": 121,
          "total": 231
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20241202_agentless-1.5_claude-3.5-sonnet-20241022"
    },
    {
      "id": "verified/20241128_SWE-Fixer_Qwen2.5-7b-retriever_Qwen2.5-72b-editor_20241128",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-11-28",
      "harness": "SWE-Fixer",
      "modelSlug": "Qwen2.5-7b-retriever_Qwen2.5-72b-editor_20241128",
      "model": "Qwen2.5 7b Retriever_Qwen2.5 72b Editor_20241128",
      "resolved": 151,
      "attempted": 153,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 30.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sympy/sympy": {
          "resolved": 16,
          "total": 75
        },
        "django/django": {
          "resolved": 81,
          "total": 231
        },
        "scikit-learn/scikit-learn": {
          "resolved": 17,
          "total": 32
        },
        "sphinx-doc/sphinx": {
          "resolved": 11,
          "total": 44
        },
        "astropy/astropy": {
          "resolved": 5,
          "total": 22
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "psf/requests": {
          "resolved": 2,
          "total": 8
        },
        "matplotlib/matplotlib": {
          "resolved": 6,
          "total": 34
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "pydata/xarray": {
          "resolved": 6,
          "total": 22
        },
        "pytest-dev/pytest": {
          "resolved": 4,
          "total": 19
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20241128_SWE-Fixer_Qwen2.5-7b-retriever_Qwen2.5-72b-editor_20241128"
    },
    {
      "id": "verified/20241120_artemis_agent",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-11-20",
      "harness": "artemis",
      "modelSlug": "agent",
      "model": "Agent",
      "resolved": 160,
      "attempted": 171,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 32,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "psf/requests": {
          "resolved": 1,
          "total": 8
        },
        "sphinx-doc/sphinx": {
          "resolved": 6,
          "total": 44
        },
        "pydata/xarray": {
          "resolved": 8,
          "total": 22
        },
        "pytest-dev/pytest": {
          "resolved": 6,
          "total": 19
        },
        "astropy/astropy": {
          "resolved": 6,
          "total": 22
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 1
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "django/django": {
          "resolved": 82,
          "total": 231
        },
        "scikit-learn/scikit-learn": {
          "resolved": 18,
          "total": 32
        },
        "sympy/sympy": {
          "resolved": 23,
          "total": 75
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        },
        "matplotlib/matplotlib": {
          "resolved": 8,
          "total": 34
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20241120_artemis_agent"
    },
    {
      "id": "verified/20241028_agentless-1.5_gpt4o",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-10-28",
      "harness": "agentless-1.5",
      "modelSlug": "gpt4o",
      "model": "Gpt4o",
      "resolved": 194,
      "attempted": 198,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 38.8,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pydata/xarray": {
          "resolved": 6,
          "total": 22
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        },
        "pytest-dev/pytest": {
          "resolved": 7,
          "total": 19
        },
        "django/django": {
          "resolved": 96,
          "total": 231
        },
        "psf/requests": {
          "resolved": 5,
          "total": 8
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "matplotlib/matplotlib": {
          "resolved": 13,
          "total": 34
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "sympy/sympy": {
          "resolved": 29,
          "total": 75
        },
        "scikit-learn/scikit-learn": {
          "resolved": 19,
          "total": 32
        },
        "sphinx-doc/sphinx": {
          "resolved": 8,
          "total": 44
        },
        "astropy/astropy": {
          "resolved": 8,
          "total": 22
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20241028_agentless-1.5_gpt4o"
    },
    {
      "id": "verified/20241025_composio_swekit",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-10-25",
      "harness": "composio",
      "modelSlug": "swekit",
      "model": "Swekit",
      "resolved": 243,
      "attempted": 244,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 48.6,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "astropy/astropy": {
          "resolved": 8,
          "total": 22
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "django/django": {
          "resolved": 119,
          "total": 231
        },
        "scikit-learn/scikit-learn": {
          "resolved": 23,
          "total": 32
        },
        "pytest-dev/pytest": {
          "resolved": 12,
          "total": 19
        },
        "pylint-dev/pylint": {
          "resolved": 4,
          "total": 10
        },
        "sympy/sympy": {
          "resolved": 39,
          "total": 75
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "sphinx-doc/sphinx": {
          "resolved": 11,
          "total": 44
        },
        "matplotlib/matplotlib": {
          "resolved": 13,
          "total": 34
        },
        "psf/requests": {
          "resolved": 3,
          "total": 8
        },
        "pydata/xarray": {
          "resolved": 10,
          "total": 22
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20241025_composio_swekit"
    },
    {
      "id": "verified/20241022_tools_claude-3-5-sonnet-updated",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-10-22",
      "harness": "tools",
      "modelSlug": "claude-3-5-sonnet-updated",
      "model": "Claude 3.5 Sonnet Updated",
      "resolved": 245,
      "attempted": 262,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 49,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "pytest-dev/pytest": {
          "resolved": 11,
          "total": 19
        },
        "matplotlib/matplotlib": {
          "resolved": 12,
          "total": 34
        },
        "psf/requests": {
          "resolved": 4,
          "total": 8
        },
        "sympy/sympy": {
          "resolved": 35,
          "total": 75
        },
        "django/django": {
          "resolved": 119,
          "total": 231
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 10
        },
        "astropy/astropy": {
          "resolved": 11,
          "total": 22
        },
        "scikit-learn/scikit-learn": {
          "resolved": 24,
          "total": 32
        },
        "pydata/xarray": {
          "resolved": 12,
          "total": 22
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "sphinx-doc/sphinx": {
          "resolved": 13,
          "total": 44
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20241022_tools_claude-3-5-sonnet-updated"
    },
    {
      "id": "verified/20241022_tools_claude-3-5-haiku",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-10-22",
      "harness": "tools",
      "modelSlug": "claude-3-5-haiku",
      "model": "Claude 3.5 Haiku",
      "resolved": 203,
      "attempted": 222,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 40.6,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "sphinx-doc/sphinx": {
          "resolved": 12,
          "total": 44
        },
        "sympy/sympy": {
          "resolved": 28,
          "total": 75
        },
        "matplotlib/matplotlib": {
          "resolved": 12,
          "total": 34
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "scikit-learn/scikit-learn": {
          "resolved": 26,
          "total": 32
        },
        "pydata/xarray": {
          "resolved": 7,
          "total": 22
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        },
        "astropy/astropy": {
          "resolved": 6,
          "total": 22
        },
        "django/django": {
          "resolved": 97,
          "total": 231
        },
        "psf/requests": {
          "resolved": 4,
          "total": 8
        },
        "pytest-dev/pytest": {
          "resolved": 8,
          "total": 19
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20241022_tools_claude-3-5-haiku"
    },
    {
      "id": "verified/20241016_composio_swekit",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-10-16",
      "harness": "composio",
      "modelSlug": "swekit",
      "model": "Swekit",
      "resolved": 203,
      "attempted": 203,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 40.6,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "matplotlib/matplotlib": {
          "resolved": 8,
          "total": 34
        },
        "sphinx-doc/sphinx": {
          "resolved": 10,
          "total": 44
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "astropy/astropy": {
          "resolved": 5,
          "total": 22
        },
        "pytest-dev/pytest": {
          "resolved": 8,
          "total": 19
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        },
        "psf/requests": {
          "resolved": 3,
          "total": 8
        },
        "sympy/sympy": {
          "resolved": 29,
          "total": 75
        },
        "django/django": {
          "resolved": 109,
          "total": 231
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "pydata/xarray": {
          "resolved": 9,
          "total": 22
        },
        "scikit-learn/scikit-learn": {
          "resolved": 19,
          "total": 32
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20241016_composio_swekit"
    },
    {
      "id": "verified/20241002_lingma-agent_lingma-swe-gpt-72b",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-10-02",
      "harness": "lingma-agent",
      "modelSlug": "lingma-swe-gpt-72b",
      "model": "Lingma Swe GPT 72b",
      "resolved": 144,
      "attempted": 153,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 28.8,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "django/django": {
          "resolved": 77,
          "total": 231
        },
        "pytest-dev/pytest": {
          "resolved": 4,
          "total": 19
        },
        "matplotlib/matplotlib": {
          "resolved": 7,
          "total": 34
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "sphinx-doc/sphinx": {
          "resolved": 11,
          "total": 44
        },
        "sympy/sympy": {
          "resolved": 12,
          "total": 75
        },
        "scikit-learn/scikit-learn": {
          "resolved": 17,
          "total": 32
        },
        "pydata/xarray": {
          "resolved": 5,
          "total": 22
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 10
        },
        "psf/requests": {
          "resolved": 4,
          "total": 8
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "astropy/astropy": {
          "resolved": 3,
          "total": 22
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20241002_lingma-agent_lingma-swe-gpt-72b"
    },
    {
      "id": "verified/20241002_lingma-agent_lingma-swe-gpt-7b",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-10-02",
      "harness": "lingma-agent",
      "modelSlug": "lingma-swe-gpt-7b",
      "model": "Lingma Swe GPT 7b",
      "resolved": 91,
      "attempted": 179,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 18.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "sphinx-doc/sphinx": {
          "resolved": 3,
          "total": 44
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 22
        },
        "psf/requests": {
          "resolved": 4,
          "total": 8
        },
        "django/django": {
          "resolved": 53,
          "total": 231
        },
        "matplotlib/matplotlib": {
          "resolved": 5,
          "total": 34
        },
        "scikit-learn/scikit-learn": {
          "resolved": 9,
          "total": 32
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 22
        },
        "sympy/sympy": {
          "resolved": 9,
          "total": 75
        },
        "pytest-dev/pytest": {
          "resolved": 3,
          "total": 19
        },
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 10
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20241002_lingma-agent_lingma-swe-gpt-7b"
    },
    {
      "id": "verified/20240918_lingma-agent_lingma-swe-gpt-72b",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-09-18",
      "harness": "lingma-agent",
      "modelSlug": "lingma-swe-gpt-72b",
      "model": "Lingma Swe GPT 72b",
      "resolved": 125,
      "attempted": 142,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 25,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pydata/xarray": {
          "resolved": 8,
          "total": 22
        },
        "django/django": {
          "resolved": 69,
          "total": 231
        },
        "pytest-dev/pytest": {
          "resolved": 4,
          "total": 19
        },
        "sympy/sympy": {
          "resolved": 10,
          "total": 75
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "sphinx-doc/sphinx": {
          "resolved": 6,
          "total": 44
        },
        "psf/requests": {
          "resolved": 3,
          "total": 8
        },
        "matplotlib/matplotlib": {
          "resolved": 10,
          "total": 34
        },
        "scikit-learn/scikit-learn": {
          "resolved": 8,
          "total": 32
        },
        "astropy/astropy": {
          "resolved": 4,
          "total": 22
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20240918_lingma-agent_lingma-swe-gpt-72b"
    },
    {
      "id": "verified/20240918_lingma-agent_lingma-swe-gpt-7b",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-09-18",
      "harness": "lingma-agent",
      "modelSlug": "lingma-swe-gpt-7b",
      "model": "Lingma Swe GPT 7b",
      "resolved": 51,
      "attempted": 92,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 10.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "psf/requests": {
          "resolved": 1,
          "total": 8
        },
        "sphinx-doc/sphinx": {
          "resolved": 2,
          "total": 44
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 22
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "scikit-learn/scikit-learn": {
          "resolved": 7,
          "total": 32
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "django/django": {
          "resolved": 26,
          "total": 231
        },
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 10
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 22
        },
        "sympy/sympy": {
          "resolved": 3,
          "total": 75
        },
        "pytest-dev/pytest": {
          "resolved": 3,
          "total": 19
        },
        "matplotlib/matplotlib": {
          "resolved": 4,
          "total": 34
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20240918_lingma-agent_lingma-swe-gpt-7b"
    },
    {
      "id": "verified/20240728_sweagent_gpt4o",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-07-28",
      "harness": "sweagent",
      "modelSlug": "gpt4o",
      "model": "Gpt4o",
      "resolved": 116,
      "attempted": 169,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 23.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "pydata/xarray": {
          "resolved": 6,
          "total": 22
        },
        "matplotlib/matplotlib": {
          "resolved": 0,
          "total": 34
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "pytest-dev/pytest": {
          "resolved": 5,
          "total": 19
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 10
        },
        "sphinx-doc/sphinx": {
          "resolved": 0,
          "total": 44
        },
        "astropy/astropy": {
          "resolved": 3,
          "total": 22
        },
        "django/django": {
          "resolved": 66,
          "total": 231
        },
        "psf/requests": {
          "resolved": 3,
          "total": 8
        },
        "sympy/sympy": {
          "resolved": 17,
          "total": 75
        },
        "scikit-learn/scikit-learn": {
          "resolved": 12,
          "total": 32
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20240728_sweagent_gpt4o"
    },
    {
      "id": "verified/20240620_sweagent_claude3.5sonnet",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-06-20",
      "harness": "sweagent",
      "modelSlug": "claude3.5sonnet",
      "model": "Claude3.5sonnet",
      "resolved": 168,
      "attempted": 182,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 33.6,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "scikit-learn/scikit-learn": {
          "resolved": 18,
          "total": 32
        },
        "sphinx-doc/sphinx": {
          "resolved": 0,
          "total": 44
        },
        "sympy/sympy": {
          "resolved": 24,
          "total": 75
        },
        "pydata/xarray": {
          "resolved": 9,
          "total": 22
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 10
        },
        "django/django": {
          "resolved": 89,
          "total": 231
        },
        "psf/requests": {
          "resolved": 4,
          "total": 8
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "pallets/flask": {
          "resolved": 0,
          "total": 1
        },
        "pytest-dev/pytest": {
          "resolved": 7,
          "total": 19
        },
        "matplotlib/matplotlib": {
          "resolved": 9,
          "total": 34
        },
        "astropy/astropy": {
          "resolved": 6,
          "total": 22
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20240620_sweagent_claude3.5sonnet"
    },
    {
      "id": "verified/20240617_factory_code_droid",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-06-17",
      "harness": "factory",
      "modelSlug": "code_droid",
      "model": "Code_droid",
      "resolved": 185,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 37,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "astropy/astropy": {
          "resolved": 9,
          "total": 95
        },
        "django/django": {
          "resolved": 97,
          "total": 850
        },
        "matplotlib/matplotlib": {
          "resolved": 9,
          "total": 184
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 11
        },
        "psf/requests": {
          "resolved": 5,
          "total": 44
        },
        "pydata/xarray": {
          "resolved": 7,
          "total": 110
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 57
        },
        "pytest-dev/pytest": {
          "resolved": 6,
          "total": 119
        },
        "scikit-learn/scikit-learn": {
          "resolved": 13,
          "total": 229
        },
        "sphinx-doc/sphinx": {
          "resolved": 10,
          "total": 187
        },
        "sympy/sympy": {
          "resolved": 25,
          "total": 386
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20240617_factory_code_droid"
    },
    {
      "id": "verified/20240615_appmap-navie_gpt4o",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-06-15",
      "harness": "appmap-navie",
      "modelSlug": "gpt4o",
      "model": "Gpt4o",
      "resolved": 131,
      "attempted": 494,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 26.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "django/django": {
          "resolved": 67,
          "total": 850
        },
        "sphinx-doc/sphinx": {
          "resolved": 7,
          "total": 187
        },
        "sympy/sympy": {
          "resolved": 17,
          "total": 386
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 11
        },
        "pydata/xarray": {
          "resolved": 5,
          "total": 110
        },
        "pytest-dev/pytest": {
          "resolved": 4,
          "total": 119
        },
        "matplotlib/matplotlib": {
          "resolved": 8,
          "total": 184
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 57
        },
        "scikit-learn/scikit-learn": {
          "resolved": 13,
          "total": 229
        },
        "astropy/astropy": {
          "resolved": 4,
          "total": 95
        },
        "psf/requests": {
          "resolved": 2,
          "total": 44
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20240615_appmap-navie_gpt4o"
    },
    {
      "id": "verified/20240612_MASAI_gpt4o",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-06-12",
      "harness": "MASAI",
      "modelSlug": "gpt4o",
      "model": "Gpt4o",
      "resolved": 163,
      "attempted": 189,
      "total": 500,
      "totalBasis": "split-constant, by-repo agrees",
      "resolvedPct": 32.6,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "astropy/astropy": {
          "resolved": 4,
          "total": 22
        },
        "sphinx-doc/sphinx": {
          "resolved": 5,
          "total": 44
        },
        "django/django": {
          "resolved": 91,
          "total": 231
        },
        "pylint-dev/pylint": {
          "resolved": 3,
          "total": 10
        },
        "psf/requests": {
          "resolved": 1,
          "total": 8
        },
        "mwaskom/seaborn": {
          "resolved": 0,
          "total": 2
        },
        "sympy/sympy": {
          "resolved": 25,
          "total": 75
        },
        "pytest-dev/pytest": {
          "resolved": 8,
          "total": 19
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 1
        },
        "scikit-learn/scikit-learn": {
          "resolved": 12,
          "total": 32
        },
        "matplotlib/matplotlib": {
          "resolved": 5,
          "total": 34
        },
        "pydata/xarray": {
          "resolved": 8,
          "total": 22
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20240612_MASAI_gpt4o"
    },
    {
      "id": "verified/20240402_sweagent_gpt4",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-04-02",
      "harness": "sweagent",
      "modelSlug": "gpt4",
      "model": "Gpt4",
      "resolved": 112,
      "attempted": 472,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 22.4,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "django/django": {
          "resolved": 60,
          "total": 850
        },
        "sympy/sympy": {
          "resolved": 13,
          "total": 386
        },
        "astropy/astropy": {
          "resolved": 6,
          "total": 95
        },
        "psf/requests": {
          "resolved": 4,
          "total": 44
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 57
        },
        "scikit-learn/scikit-learn": {
          "resolved": 10,
          "total": 229
        },
        "matplotlib/matplotlib": {
          "resolved": 7,
          "total": 184
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 110
        },
        "pytest-dev/pytest": {
          "resolved": 5,
          "total": 119
        },
        "sphinx-doc/sphinx": {
          "resolved": 4,
          "total": 187
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20240402_sweagent_gpt4"
    },
    {
      "id": "verified/20240402_sweagent_claude3opus",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-04-02",
      "harness": "sweagent",
      "modelSlug": "claude3opus",
      "model": "Claude3opus",
      "resolved": 79,
      "attempted": 455,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 15.8,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "scikit-learn/scikit-learn": {
          "resolved": 13,
          "total": 229
        },
        "django/django": {
          "resolved": 47,
          "total": 850
        },
        "pytest-dev/pytest": {
          "resolved": 4,
          "total": 119
        },
        "matplotlib/matplotlib": {
          "resolved": 4,
          "total": 184
        },
        "astropy/astropy": {
          "resolved": 4,
          "total": 95
        },
        "sympy/sympy": {
          "resolved": 12,
          "total": 386
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 57
        },
        "psf/requests": {
          "resolved": 1,
          "total": 44
        },
        "pydata/xarray": {
          "resolved": 3,
          "total": 110
        },
        "pallets/flask": {
          "resolved": 1,
          "total": 11
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20240402_sweagent_claude3opus"
    },
    {
      "id": "verified/20240402_rag_claude3opus",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-04-02",
      "harness": "rag",
      "modelSlug": "claude3opus",
      "model": "Claude3opus",
      "resolved": 35,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 7,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "sympy/sympy": {
          "resolved": 4,
          "total": 386
        },
        "astropy/astropy": {
          "resolved": 2,
          "total": 95
        },
        "pytest-dev/pytest": {
          "resolved": 1,
          "total": 119
        },
        "scikit-learn/scikit-learn": {
          "resolved": 6,
          "total": 229
        },
        "django/django": {
          "resolved": 20,
          "total": 850
        },
        "pylint-dev/pylint": {
          "resolved": 2,
          "total": 57
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20240402_rag_claude3opus"
    },
    {
      "id": "verified/20240402_rag_gpt4",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2024-04-02",
      "harness": "rag",
      "modelSlug": "gpt4",
      "model": "Gpt4",
      "resolved": 14,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 2.8,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "django/django": {
          "resolved": 11,
          "total": 850
        },
        "scikit-learn/scikit-learn": {
          "resolved": 1,
          "total": 229
        },
        "astropy/astropy": {
          "resolved": 1,
          "total": 95
        },
        "psf/requests": {
          "resolved": 1,
          "total": 44
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20240402_rag_gpt4"
    },
    {
      "id": "verified/20231010_rag_claude2",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2023-10-10",
      "harness": "rag",
      "modelSlug": "claude2",
      "model": "Claude2",
      "resolved": 22,
      "attempted": 499,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 4.4,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "pytest-dev/pytest": {
          "resolved": 1,
          "total": 119
        },
        "django/django": {
          "resolved": 14,
          "total": 850
        },
        "scikit-learn/scikit-learn": {
          "resolved": 5,
          "total": 229
        },
        "pylint-dev/pylint": {
          "resolved": 1,
          "total": 57
        },
        "psf/requests": {
          "resolved": 1,
          "total": 44
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20231010_rag_claude2"
    },
    {
      "id": "verified/20231010_rag_swellama7b",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2023-10-10",
      "harness": "rag",
      "modelSlug": "swellama7b",
      "model": "Swellama7b",
      "resolved": 7,
      "attempted": 494,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 1.4,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "django/django": {
          "resolved": 4,
          "total": 850
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 110
        },
        "pytest-dev/pytest": {
          "resolved": 1,
          "total": 119
        },
        "sympy/sympy": {
          "resolved": 1,
          "total": 386
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20231010_rag_swellama7b"
    },
    {
      "id": "verified/20231010_rag_swellama13b",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2023-10-10",
      "harness": "rag",
      "modelSlug": "swellama13b",
      "model": "Swellama13b",
      "resolved": 6,
      "attempted": 479,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 1.2,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "django/django": {
          "resolved": 2,
          "total": 850
        },
        "pydata/xarray": {
          "resolved": 1,
          "total": 110
        },
        "pytest-dev/pytest": {
          "resolved": 1,
          "total": 119
        },
        "sympy/sympy": {
          "resolved": 2,
          "total": 386
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20231010_rag_swellama13b"
    },
    {
      "id": "verified/20231010_rag_gpt35",
      "suite": "SWE-bench Verified",
      "split": "verified",
      "date": "2023-10-10",
      "harness": "rag",
      "modelSlug": "gpt35",
      "model": "Gpt35",
      "resolved": 2,
      "attempted": 500,
      "total": 500,
      "totalBasis": "split-constant",
      "resolvedPct": 0.4,
      "totalCostUsd": null,
      "meanCostUsd": null,
      "totalApiCalls": null,
      "byRepo": {
        "scikit-learn/scikit-learn": {
          "resolved": 1,
          "total": 229
        },
        "django/django": {
          "resolved": 1,
          "total": 850
        }
      },
      "sourceUrl": "https://github.com/SWE-bench/experiments/tree/main/evaluation/verified/20231010_rag_gpt35"
    }
  ],
  "flagged": [
    {
      "id": "verified/20260226_mini-v2.0.0_gemini-3-pro-high",
      "model": "Gemini 3 Pro",
      "suite": "SWE-bench Verified",
      "resolvedPct": 0,
      "totalCostUsd": 480.01,
      "reason": "zero resolved with non-trivial spend — treated as a failed harness run, not a model score"
    }
  ]
}
