{
  "generated_at": "2026-10-04 02:45Z",
  "status": "FINAL",
  "window": {
    "first_merge": "2026-09-05T04:01:42Z",
    "last_merge": "2026-10-04T02:34:22Z",
    "days": 28
  },
  "cohorts": {
    "shipper": {
      "merged": 136,
      "additions": 38182,
      "deletions": 8620,
      "size_median": 192.5
    },
    "human": {
      "merged": 66,
      "additions": 49703,
      "deletions": 12125,
      "size_median": 91.5
    }
  },
  "autonomy": {
    "loop": {
      "prs": 50,
      "share": 0.36764705882352944,
      "share_ci": [
        0.2912843697911778,
        0.45128150474115014
      ],
      "hours_open_median": 0.0011111111111111111
    },
    "hand": {
      "prs": 66,
      "share": 0.4852941176470588,
      "share_ci": [
        0.40286095897691965,
        0.56853524930445
      ],
      "hours_open_median": 20.80513888888889
    },
    "unrecorded": {
      "prs": 20,
      "share": 0.14705882352941177,
      "share_ci": [
        0.09725850554788583,
        0.2162504932049888
      ],
      "hours_open_median": 0.0020833333333333333
    }
  },
  "baseline": {
    "human_prs": 66,
    "reverted": 0
  },
  "escapes": {
    "shipper_prs": 136,
    "reverted": 0,
    "reverted_rate_ci": [
      0.0,
      0.02747108156657246
    ],
    "reported_candidates": 2,
    "reported_confirmed": 0,
    "main_red_events": 2,
    "main_red_real": 1,
    "main_red_flake": 1,
    "post_merge_rulings": 4,
    "post_merge_accepted_from_agent": 4
  },
  "funnel": {
    "distinct_issues": 221,
    "host_blocked_only": 10,
    "actionable": 211,
    "merged": 50,
    "reached_a_pr": 104,
    "merged_rate": 0.23696682464454977,
    "merged_rate_ci": [
      0.18461523116737097,
      0.2987250530234003
    ],
    "attempts_per_merged_median": 1.0,
    "first_drive_to_merge_hours_median": 4.7025,
    "attempts_total": 1297,
    "by_best_outcome": {
      "merged": 50,
      "pr_open": 54,
      "needs_human": 42,
      "declined": 57,
      "failed": 7,
      "host_blocked": 10,
      "other": 1
    }
  },
  "audit": {
    "sample": {
      "n": 30,
      "audited": 30,
      "population": 130,
      "seed": 7,
      "drawn_at": "2026-10-03T18:22:07Z",
      "added_since": 6,
      "strata": {
        "hand": 66,
        "loop": 44,
        "unrecorded": 20
      },
      "allocation": {
        "hand": 15,
        "loop": 10,
        "unrecorded": 5
      }
    },
    "auditor": {
      "model": "gpt-5.5",
      "effort": "high"
    },
    "tally": {
      "resolves": {
        "yes": 25,
        "partial": 5
      },
      "defect": {
        "none": 25,
        "minor": 2,
        "major": 3
      },
      "scope": {
        "tight": 30
      },
      "tests": {
        "meaningful": 29,
        "weak": 1
      },
      "confidence": {
        "high": 28,
        "medium": 2
      }
    },
    "flagged": 8,
    "clean": 22,
    "inconclusive": 0,
    "flagged_by_route": {
      "hand": 5,
      "loop": 1,
      "unrecorded": 2
    },
    "audited_by_route": {
      "hand": 15,
      "loop": 10,
      "unrecorded": 5
    },
    "flagged_real": 5,
    "flagged_false": 3,
    "spot_checked": 8,
    "spot_misses": 0,
    "rulings_by_person": 16,
    "rulings_accepted_from_agent": 16,
    "precision": {
      "k": 5,
      "n": 8,
      "ci": [
        0.30573785458380187,
        0.863158240538479
      ]
    },
    "miss_rate": {
      "k": 0,
      "n": 8,
      "ci": [
        0.0,
        0.3244156195108769
      ]
    },
    "kappa": 0.625,
    "kappa_n": 16,
    "confirmed_problem_prs": 5,
    "confirmed_problem_ci": [
      0.07336434240351683,
      0.33564705185680177
    ],
    "by_week": [
      {
        "week": 1,
        "start": "2026-09-05",
        "n": 19,
        "confirmed": 5,
        "ci": [
          0.11806238339540162,
          0.48791968441070643
        ]
      },
      {
        "week": 2,
        "start": "2026-09-12",
        "n": 5,
        "confirmed": 0,
        "ci": [
          0.0,
          0.43449149475208104
        ]
      },
      {
        "week": 4,
        "start": "2026-09-26",
        "n": 5,
        "confirmed": 0,
        "ci": [
          0.0,
          0.43449149475208104
        ]
      },
      {
        "week": 5,
        "start": "2026-10-03",
        "n": 1,
        "confirmed": 0,
        "ci": [
          0.0,
          0.7934567085261071
        ]
      }
    ]
  },
  "cost": {
    "merged_prs": 136,
    "with_usage_on_disk": 104,
    "coverage": 0.7647058823529411,
    "issues_with_usage": 167,
    "tokens_per_merged_pr": {
      "n": 104,
      "median": 1470654.5,
      "mean": 3368483.980769231,
      "max": 28139852.0
    },
    "dollars_per_merged_pr": {
      "n": 104,
      "median": 1.49876325,
      "mean": 2.9797120961538464,
      "max": 20.94774375,
      "priced": 104
    },
    "total_tokens": 499347884,
    "total_dollars": 445.76425344999996,
    "share_tokens_on_merged": 0.6947759490255495,
    "share_tokens_on_unmerged": 0.3052240509744505,
    "models": [
      "claude-opus-5",
      "claude-opus-5-5",
      "gpt-5.5"
    ],
    "prices_retrieved": "2026-10-03"
  },
  "claims": [
    "Built an autonomous issue-to-merged-PR pipeline (Claude authors, Codex reviews adversarially, a third model can veto; every gate fails closed). In 28 days it authored 136 merged pull requests on a production codebase; 50 (37%) were merged by the pipeline itself with no human action, and the rest waited a median of 21 hours for a person.",
    "Measured outcomes, not activity: 0 of 136 merged PRs were reverted (95% CI 0-3%); a post-merge gate on the base branch caught 1 real semantic-merge failure(s) and 1 flake(s) (4 of the 4 post-merge rulings accepted from an agent's written reasoning).",
    "Ran a blind cross-model audit (Codex reading what Claude wrote, without the pipeline's verdicts) of a seeded random sample of 30 merged PRs of the 130 PRs merged when it was drawn (6 merged since are not covered): it flagged 8; after human review 5 of 8 flags were real defects (62%, 95% CI 31-86%) (16 of the 16 rulings accepted from an agent's written reasoning, not independently re-derived), 5 of 30 PRs confirmed to have a real problem (95% CI 7-34%).",
    "Rebuilt per-issue cost from the agents' own session logs: a median $1.50 API-equivalent per merged PR (n=104, an estimate on flat plans at list prices retrieved 2026-10-03), with 31% of tokens spent on work that never merged."
  ],
  "limits": [
    "One repository and one maintainer. Nothing here says how the pipeline behaves on code it was not built around.",
    "A defect nobody has noticed is not in the post-merge numbers; the blind audit exists to estimate those, on a sample of 30, which is small. Intervals are Wilson 95%.",
    "The auditor is one model family (Codex). It over-flags, which is why every flag was ruled on by a person, and it can miss problems, which is why a random sample of its clean verdicts was checked.",
    "4 of the 4 post-merge rulings (whether a red main or a later issue was a real defect) were signed by a person who accepted an agent's written reasoning rather than re-deriving it.",
    "16 of the 16 audit rulings were signed by a person who accepted an agent's written reasoning rather than re-deriving it from the code. The agent verified each against the code at the merge commit, but 'ruled by a person' here means the person took responsibility for it, not that they redid it.",
    "The audit sampled the 130 PRs merged when it was drawn; 6 merged since are not covered by it.",
    "\"136 PRs\" counts what the pipeline authored and gated; 66 of them were merged by a person or an agent working at a person's direction, not by the pipeline.",
    "Cost covers 76% of merged PRs (older logs are pruned), counts tokens exactly and dollars as an API-equivalent estimate (both agents are on flat plans), and omits the veto gate, which runs in a throwaway directory."
  ]
}