{
  "schema": "agent-public-bench@1",
  "generatedAt": "2026-10-07T00:00:00.000Z",
  "url": "https://agent.sasid.ai/benchmarks/blind-review-head-to-head",
  "study": {
    "slug": "blind-review-head-to-head",
    "title": "AI pull requests vs merged human pull requests, judged blind",
    "seoTitle": "AI vs human pull requests: a blind multi-model review",
    "description": "A blind panel of Claude and GPT critics preferred Agent's change over the merged human change on 9 of 12 real tasks. Votes, scores, caveats.",
    "question": "When critics cannot see which change came from a person, do they prefer the AI worker’s pull request or the one the maintainers merged?",
    "answer": "On the latest attempt per task, the blind panel preferred the AI change on 9 of 12 tasks (75%, 95% interval 47% to 91%). On the first scored attempt it was 6 of 12 (25% to 75%). Later attempts learned from earlier ones, so the first-attempt figure is the cleaner estimate. Public OSS tasks: 4 of 5; private Go tasks: 5 of 7. Across 132 single verdicts, 69% preferred the AI change. A critic panel's preference is not a merge decision and not a correctness proof.",
    "date": "2026-09-28",
    "updated": "2026-10-05",
    "tags": [
      "code-review",
      "ai-vs-human",
      "llm-as-judge",
      "pull-requests"
    ],
    "method": [
      "Each task is a real merged pull request: the issue as the ticket, the repository at the base commit, the merged change as the human reference.",
      "The Agent worker solves the ticket in a sandbox without seeing the reference. Its pull request and the merged one form a pair.",
      "Critic models (Claude Opus 5, Claude Fable 5, Claude Sonnet, and on some pairs Codex or GPT 5.5) review both changes without labels, once in each order.",
      "Decision rule 2: the AI change wins on a strict majority of verdicts and no panel veto (serious issues flagged by at least two critics).",
      "Every scored attempt is kept. A re-scoring of the same pair under the current rule replaces the earlier scoring."
    ],
    "caveats": [
      "n = 12 tasks: intervals are wide.",
      "Most critics are Anthropic models, and the worker runs on an Anthropic model. Same-family preference is possible; the per-critic chart shows the OpenAI critics’ share.",
      "Latest attempts are not independent first tries: they had lessons, review replays and some operator answers from earlier attempts on the same task.",
      "7 of the 12 tasks come from one private Go service. They carry neutral labels (private-go-a and so on), and only their language and kind are published.",
      "4 pairs were flagged for position bias (a critic flipped when the order swapped).",
      "The human change was merged, reviewed and shipped. A critic preference says nothing about long-term maintenance cost."
    ],
    "sourceIds": [
      "agent-blind-review"
    ],
    "stats": [
      {
        "id": "ai-preferred-latest",
        "label": "Tasks where the panel preferred the AI change (latest attempt)",
        "value": 0.75,
        "unit": "rate",
        "display": "75% (9/12)",
        "n": 12,
        "ci": [
          0.4677,
          0.9111
        ],
        "note": "Latest attempts come after earlier attempts on the same task; see caveats."
      },
      {
        "id": "ai-preferred-first",
        "label": "Tasks where the panel preferred the AI change (first scored attempt)",
        "value": 0.5,
        "unit": "rate",
        "display": "50% (6/12)",
        "n": 12,
        "ci": [
          0.2538,
          0.7462
        ]
      },
      {
        "id": "ai-preferred-all-pairs",
        "label": "All scored pairs where the panel preferred the AI change",
        "value": 0.6,
        "unit": "rate",
        "display": "60% (12/20)",
        "n": 20,
        "ci": [
          0.3866,
          0.7812
        ]
      },
      {
        "id": "ai-preferred-public",
        "label": "Public OSS tasks, latest attempt",
        "value": 0.8,
        "unit": "rate",
        "display": "80% (4/5)",
        "n": 5,
        "ci": [
          0.3755,
          0.9638
        ]
      },
      {
        "id": "verdicts-ai",
        "label": "Single critic verdicts that preferred the AI change",
        "value": 0.6894,
        "unit": "rate",
        "display": "69% (91/132)",
        "n": 132,
        "ci": [
          0.606,
          0.762
        ]
      },
      {
        "id": "review-spend",
        "label": "Notional spend: worker runs + critic panel",
        "value": 1310.5,
        "unit": "usd",
        "display": "$979 + $332",
        "n": 20,
        "note": "List-price estimates of subscription calls over all scored pairs."
      },
      {
        "id": "position-bias-flags",
        "label": "Pairs flagged for position bias",
        "value": 4,
        "unit": "count",
        "display": "4 of 20",
        "n": 20,
        "note": "A critic model flipped its preference when the two changes swapped places."
      }
    ],
    "charts": [
      {
        "id": "blind-review-first-vs-latest",
        "title": "First attempt vs latest attempt",
        "subtitle": "Share of tasks where the blind panel preferred the AI change",
        "kind": "dot-range",
        "unit": "rate",
        "yLabel": "Tasks preferred",
        "series": [
          {
            "name": "AI preferred",
            "points": [
              {
                "label": "First scored attempt",
                "value": 0.5,
                "lo": 0.2538,
                "hi": 0.7462,
                "n": 12
              },
              {
                "label": "Latest attempt",
                "value": 0.75,
                "lo": 0.4677,
                "hi": 0.9111,
                "n": 12,
                "highlight": true
              },
              {
                "label": "Public OSS tasks, latest",
                "value": 0.8,
                "lo": 0.3755,
                "hi": 0.9638,
                "n": 5
              },
              {
                "label": "Private tasks, latest",
                "value": 0.7143,
                "lo": 0.3589,
                "hi": 0.9178,
                "n": 7
              }
            ]
          }
        ],
        "note": "Later attempts had the earlier attempts’ lessons, review replays and, on some tasks, operator answers. They are not independent first tries.",
        "sourceIds": [
          "agent-blind-review"
        ]
      },
      {
        "id": "blind-review-votes-by-task",
        "title": "Blind panel votes per task (latest attempt)",
        "subtitle": "Each critic judges both orders without labels",
        "kind": "stacked-bar",
        "unit": "count",
        "yLabel": "Verdicts",
        "series": [
          {
            "name": "Prefers AI change",
            "points": [
              {
                "label": "h3js/h3#1533",
                "value": 8,
                "n": 8,
                "highlight": true
              },
              {
                "label": "private-go-a (private Go)",
                "value": 8,
                "n": 8,
                "highlight": true
              },
              {
                "label": "open-telemetry/opentelemetry-go#8706",
                "value": 8,
                "n": 8,
                "highlight": true
              },
              {
                "label": "fastify/session#348",
                "value": 8,
                "n": 8,
                "highlight": true
              },
              {
                "label": "private-go-d (private Go)",
                "value": 6,
                "n": 6,
                "highlight": true
              },
              {
                "label": "redis/go-redis#3914",
                "value": 8,
                "n": 8,
                "highlight": true
              },
              {
                "label": "private-go-g (private Go)",
                "value": 6,
                "n": 6,
                "highlight": true
              },
              {
                "label": "private-go-f (private Go)",
                "value": 6,
                "n": 6,
                "highlight": true
              },
              {
                "label": "private-go-e (private Go)",
                "value": 6,
                "n": 6,
                "highlight": true
              },
              {
                "label": "aio-libs/aiohttp#13122",
                "value": 4,
                "n": 8,
                "highlight": false
              },
              {
                "label": "private-go-c (private Go)",
                "value": 0,
                "n": 6,
                "highlight": false
              },
              {
                "label": "private-go-b (private Go)",
                "value": 0,
                "n": 6,
                "highlight": false
              }
            ]
          },
          {
            "name": "Prefers human change",
            "points": [
              {
                "label": "h3js/h3#1533",
                "value": 0,
                "n": 8
              },
              {
                "label": "private-go-a (private Go)",
                "value": 0,
                "n": 8
              },
              {
                "label": "open-telemetry/opentelemetry-go#8706",
                "value": 0,
                "n": 8
              },
              {
                "label": "fastify/session#348",
                "value": 0,
                "n": 8
              },
              {
                "label": "private-go-d (private Go)",
                "value": 0,
                "n": 6
              },
              {
                "label": "redis/go-redis#3914",
                "value": 0,
                "n": 8
              },
              {
                "label": "private-go-g (private Go)",
                "value": 0,
                "n": 6
              },
              {
                "label": "private-go-f (private Go)",
                "value": 0,
                "n": 6
              },
              {
                "label": "private-go-e (private Go)",
                "value": 0,
                "n": 6
              },
              {
                "label": "aio-libs/aiohttp#13122",
                "value": 4,
                "n": 8
              },
              {
                "label": "private-go-c (private Go)",
                "value": 6,
                "n": 6
              },
              {
                "label": "private-go-b (private Go)",
                "value": 6,
                "n": 6
              }
            ]
          },
          {
            "name": "Tie",
            "points": [
              {
                "label": "h3js/h3#1533",
                "value": 0,
                "n": 8
              },
              {
                "label": "private-go-a (private Go)",
                "value": 0,
                "n": 8
              },
              {
                "label": "open-telemetry/opentelemetry-go#8706",
                "value": 0,
                "n": 8
              },
              {
                "label": "fastify/session#348",
                "value": 0,
                "n": 8
              },
              {
                "label": "private-go-d (private Go)",
                "value": 0,
                "n": 6
              },
              {
                "label": "redis/go-redis#3914",
                "value": 0,
                "n": 8
              },
              {
                "label": "private-go-g (private Go)",
                "value": 0,
                "n": 6
              },
              {
                "label": "private-go-f (private Go)",
                "value": 0,
                "n": 6
              },
              {
                "label": "private-go-e (private Go)",
                "value": 0,
                "n": 6
              },
              {
                "label": "aio-libs/aiohttp#13122",
                "value": 0,
                "n": 8
              },
              {
                "label": "private-go-c (private Go)",
                "value": 0,
                "n": 6
              },
              {
                "label": "private-go-b (private Go)",
                "value": 0,
                "n": 6
              }
            ]
          }
        ],
        "note": "The human change is the one the maintainers merged upstream. Private tasks are from one private Go service and carry neutral labels.",
        "sourceIds": [
          "agent-blind-review"
        ]
      },
      {
        "id": "blind-review-dimension-scores",
        "title": "What the critics scored higher",
        "subtitle": "Mean 1-5 score per dimension, latest attempt of 12 tasks",
        "kind": "grouped-bar",
        "unit": "score",
        "yLabel": "Mean score (1-5)",
        "series": [
          {
            "name": "AI change",
            "points": [
              {
                "label": "Correctness",
                "value": 4.4,
                "n": 12,
                "highlight": true
              },
              {
                "label": "Root cause",
                "value": 4.46,
                "n": 12,
                "highlight": true
              },
              {
                "label": "Edge cases",
                "value": 3.95,
                "n": 12,
                "highlight": true
              },
              {
                "label": "Tests",
                "value": 4.51,
                "n": 12,
                "highlight": true
              },
              {
                "label": "Completeness",
                "value": 4.22,
                "n": 12,
                "highlight": true
              },
              {
                "label": "Compatibility",
                "value": 4.39,
                "n": 12,
                "highlight": true
              },
              {
                "label": "Security",
                "value": 4.38,
                "n": 12,
                "highlight": true
              },
              {
                "label": "Maintainability",
                "value": 4.19,
                "n": 12,
                "highlight": true
              },
              {
                "label": "Simplicity",
                "value": 4.19,
                "n": 12,
                "highlight": true
              },
              {
                "label": "Conventions",
                "value": 4.36,
                "n": 12,
                "highlight": true
              },
              {
                "label": "Docs",
                "value": 4.04,
                "n": 12,
                "highlight": true
              },
              {
                "label": "Communication",
                "value": 4.59,
                "n": 12,
                "highlight": true
              },
              {
                "label": "Production readiness",
                "value": 4.03,
                "n": 12,
                "highlight": true
              },
              {
                "label": "Overall",
                "value": 4.14,
                "n": 12,
                "highlight": true
              }
            ]
          },
          {
            "name": "Merged human change",
            "points": [
              {
                "label": "Correctness",
                "value": 3.86,
                "n": 12
              },
              {
                "label": "Root cause",
                "value": 3.97,
                "n": 12
              },
              {
                "label": "Edge cases",
                "value": 3.49,
                "n": 12
              },
              {
                "label": "Tests",
                "value": 3.29,
                "n": 12
              },
              {
                "label": "Completeness",
                "value": 3.34,
                "n": 12
              },
              {
                "label": "Compatibility",
                "value": 3.74,
                "n": 12
              },
              {
                "label": "Security",
                "value": 4.14,
                "n": 12
              },
              {
                "label": "Maintainability",
                "value": 3.64,
                "n": 12
              },
              {
                "label": "Simplicity",
                "value": 3.85,
                "n": 12
              },
              {
                "label": "Conventions",
                "value": 3.52,
                "n": 12
              },
              {
                "label": "Docs",
                "value": 3.2,
                "n": 12
              },
              {
                "label": "Communication",
                "value": 3.58,
                "n": 12
              },
              {
                "label": "Production readiness",
                "value": 3.21,
                "n": 12
              },
              {
                "label": "Overall",
                "value": 3.2,
                "n": 12
              }
            ]
          }
        ],
        "note": "Unweighted mean over tasks of each pair’s mean critic score. Critics were not told which change came from a person.",
        "sourceIds": [
          "agent-blind-review"
        ]
      },
      {
        "id": "blind-review-critic-agreement",
        "title": "Does the judge’s model family matter?",
        "subtitle": "Share of single verdicts preferring the AI change, per critic model, all scored pairs",
        "kind": "dot-range",
        "unit": "rate",
        "yLabel": "Prefers AI change",
        "series": [
          {
            "name": "Critic model",
            "points": [
              {
                "label": "GPT 5.5 (OpenAI)",
                "value": 1,
                "lo": 0.3424,
                "hi": 1,
                "n": 2
              },
              {
                "label": "Codex (GPT) (OpenAI)",
                "value": 0.8,
                "lo": 0.4902,
                "hi": 0.9433,
                "n": 10
              },
              {
                "label": "Claude Opus 5 (Anthropic)",
                "value": 0.7,
                "lo": 0.5457,
                "hi": 0.8193,
                "n": 40
              },
              {
                "label": "Claude Sonnet (Anthropic)",
                "value": 0.675,
                "lo": 0.5202,
                "hi": 0.7992,
                "n": 40
              },
              {
                "label": "Claude Fable 5 (Anthropic)",
                "value": 0.65,
                "lo": 0.4951,
                "hi": 0.7787,
                "n": 40
              }
            ]
          }
        ],
        "note": "Whiskers are 95% Wilson intervals over single verdicts (verdicts within one pair are not independent). The worker model is Anthropic; OpenAI critics joined later and judged fewer pairs.",
        "sourceIds": [
          "agent-blind-review"
        ]
      }
    ],
    "tables": [
      {
        "id": "blind-review-pairs",
        "title": "Every scored pair",
        "columns": [
          {
            "key": "task",
            "label": "Task",
            "unit": "text"
          },
          {
            "key": "language",
            "label": "Language",
            "unit": "text"
          },
          {
            "key": "category",
            "label": "Kind",
            "unit": "text"
          },
          {
            "key": "iteration",
            "label": "Attempt",
            "unit": "count"
          },
          {
            "key": "outcome",
            "label": "Panel decision",
            "unit": "text"
          },
          {
            "key": "votes",
            "label": "Votes AI-human",
            "unit": "text"
          },
          {
            "key": "critics",
            "label": "Critics",
            "unit": "text"
          },
          {
            "key": "answers",
            "label": "Operator answers",
            "unit": "count"
          },
          {
            "key": "runUsd",
            "label": "Run cost (notional)",
            "unit": "usd"
          }
        ],
        "rows": [
          {
            "task": "aio-libs/aiohttp#13122",
            "language": "Python",
            "category": "feature",
            "iteration": 3,
            "outcome": "Human preferred",
            "votes": "4-4",
            "critics": "Claude Opus 5, Codex (GPT), Claude Fable 5, Claude Sonnet",
            "answers": 5,
            "runUsd": 78.28
          },
          {
            "task": "redis/go-redis#3914",
            "language": "Go",
            "category": "ambiguous",
            "iteration": 1,
            "outcome": "Human preferred",
            "votes": "1-5",
            "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
            "answers": 0,
            "runUsd": 21.05
          },
          {
            "task": "redis/go-redis#3914",
            "language": "Go",
            "category": "ambiguous",
            "iteration": 2,
            "outcome": "AI preferred",
            "votes": "8-0",
            "critics": "Claude Opus 5, Codex (GPT), Claude Fable 5, Claude Sonnet",
            "answers": 1,
            "runUsd": 31.5
          },
          {
            "task": "h3js/h3#1533",
            "language": "TypeScript",
            "category": "refactor",
            "iteration": 2,
            "outcome": "AI preferred",
            "votes": "8-0",
            "critics": "Claude Opus 5, Codex (GPT), Claude Fable 5, Claude Sonnet",
            "answers": 0,
            "runUsd": 21.77
          },
          {
            "task": "open-telemetry/opentelemetry-go#8706",
            "language": "Go",
            "category": "bug",
            "iteration": 1,
            "outcome": "Human preferred",
            "votes": "3-3",
            "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
            "answers": 2,
            "runUsd": 35.17
          },
          {
            "task": "open-telemetry/opentelemetry-go#8706",
            "language": "Go",
            "category": "bug",
            "iteration": 2,
            "outcome": "AI preferred",
            "votes": "8-0",
            "critics": "Claude Opus 5, Codex (GPT), Claude Fable 5, Claude Sonnet",
            "answers": 3,
            "runUsd": 80.48
          },
          {
            "task": "private-go-a (private Go)",
            "language": "Go",
            "category": "bug",
            "iteration": 1,
            "outcome": "AI preferred",
            "votes": "6-0",
            "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
            "answers": 1,
            "runUsd": 72.29
          },
          {
            "task": "private-go-a (private Go)",
            "language": "Go",
            "category": "bug",
            "iteration": 3,
            "outcome": "AI preferred",
            "votes": "6-0",
            "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
            "answers": 3,
            "runUsd": 54.51
          },
          {
            "task": "private-go-a (private Go)",
            "language": "Go",
            "category": "bug",
            "iteration": 7,
            "outcome": "AI preferred",
            "votes": "8-0",
            "critics": "Claude Opus 5, Claude Fable 5, GPT 5.5, Claude Sonnet",
            "answers": 0,
            "runUsd": 47.65
          },
          {
            "task": "private-go-b (private Go)",
            "language": "Go",
            "category": "bug",
            "iteration": 1,
            "outcome": "Human preferred",
            "votes": "1-5",
            "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
            "answers": 0,
            "runUsd": 74.97
          },
          {
            "task": "private-go-b (private Go)",
            "language": "Go",
            "category": "bug",
            "iteration": 2,
            "outcome": "Human preferred",
            "votes": "0-6",
            "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
            "answers": 2,
            "runUsd": 74.76
          },
          {
            "task": "private-go-b (private Go)",
            "language": "Go",
            "category": "bug",
            "iteration": 3,
            "outcome": "Human preferred",
            "votes": "0-6",
            "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
            "answers": 4,
            "runUsd": 70.28
          },
          {
            "task": "private-go-c (private Go)",
            "language": "Go",
            "category": "bug",
            "iteration": 1,
            "outcome": "Human preferred",
            "votes": "0-6",
            "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
            "answers": 3,
            "runUsd": 73.13
          },
          {
            "task": "private-go-d (private Go)",
            "language": "Go",
            "category": "refactor",
            "iteration": 1,
            "outcome": "AI preferred",
            "votes": "6-0",
            "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
            "answers": 3,
            "runUsd": 43.39
          },
          {
            "task": "private-go-e (private Go)",
            "language": "Go",
            "category": "feature",
            "iteration": 1,
            "outcome": "Human preferred",
            "votes": "0-6",
            "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
            "answers": 2,
            "runUsd": 22.93
          },
          {
            "task": "private-go-e (private Go)",
            "language": "Go",
            "category": "feature",
            "iteration": 2,
            "outcome": "AI preferred",
            "votes": "6-0",
            "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
            "answers": 1,
            "runUsd": 32.75
          },
          {
            "task": "private-go-e (private Go)",
            "language": "Go",
            "category": "feature",
            "iteration": 3,
            "outcome": "AI preferred",
            "votes": "6-0",
            "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
            "answers": 3,
            "runUsd": 26.81
          },
          {
            "task": "private-go-f (private Go)",
            "language": "Go",
            "category": "feature",
            "iteration": 1,
            "outcome": "AI preferred",
            "votes": "6-0",
            "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
            "answers": 2,
            "runUsd": 53.93
          },
          {
            "task": "private-go-g (private Go)",
            "language": "Go",
            "category": "feature",
            "iteration": 2,
            "outcome": "AI preferred",
            "votes": "6-0",
            "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
            "answers": 1,
            "runUsd": 41.35
          },
          {
            "task": "fastify/session#348",
            "language": "JavaScript",
            "category": "security",
            "iteration": 1,
            "outcome": "AI preferred",
            "votes": "8-0",
            "critics": "Claude Opus 5, Codex (GPT), Claude Fable 5, Claude Sonnet",
            "answers": 0,
            "runUsd": 21.53
          }
        ]
      }
    ]
  },
  "sources": [
    {
      "id": "agent-blind-review",
      "title": "Blind review panel: AI worker change vs merged human change",
      "kind": "run",
      "date": "2026-09-28",
      "note": "Each pair is judged by 3 or 4 critic models without labels, in both orders. 5 public OSS tasks and 7 private tasks. Private task names are replaced by neutral labels.",
      "data": [
        "/benchmarks/raw/blind-review/attempts.json"
      ]
    }
  ]
}
