{
  "title": "AI Undetectable Humanizer Benchmark 2026",
  "experimentDate": "2026-07-11",
  "publishedDate": "2026-08-13",
  "purpose": "Compare two humanizer candidates on detector scores and factual-quality preservation using an untouched holdout.",
  "methodology": {
    "developmentTopics": 4,
    "holdoutTopics": 8,
    "repeatsPerModelAndTopic": 2,
    "holdoutOutputsPerCandidate": 16,
    "totalHoldoutOutputs": 32,
    "detectors": [
      "ZeroGPT",
      "AI Detector"
    ],
    "baselineRequirement": "At least 90% AI on ZeroGPT",
    "detectorPassRequirement": "Below 50% AI on every available numeric detector",
    "qualityGate": {
      "minimumFactRecall": 0.95,
      "alteredMaterialFactsAllowed": 0,
      "unsupportedMaterialClaimsAllowed": 0,
      "minimumMeaningPreserved": 4,
      "minimumReadability": 4,
      "minimumNaturalness": 4,
      "minimumGenreFit": 4,
      "scoreScale": "1 to 5",
      "minimumLengthRatio": 0.7,
      "maximumLengthRatio": 1.3
    }
  },
  "baseline": {
    "samples": 12,
    "zerogptMinimum": 93.3,
    "zerogptMaximum": 100,
    "zerogptAverage": 99.35,
    "samplesBelow90": 0
  },
  "frozenHoldout": [
    {
      "candidate": "Lower-cost finalist",
      "model": "gpt-5.6-luna",
      "outputs": 16,
      "qualityPassRate": 0.5625,
      "averageFactRecall": 0.9823214286,
      "averageZeroGptAiScore": 41.08125,
      "averageAiDetectorAiScore": 88.8125,
      "bothDetectorAndQualityStrictPassRate": 0
    },
    {
      "candidate": "Selected Stealth candidate",
      "model": "gpt-5.6-terra",
      "outputs": 16,
      "qualityPassRate": 0.9375,
      "averageFactRecall": 1,
      "averageZeroGptAiScore": 46.39375,
      "averageAiDetectorAiScore": 83.9375,
      "bothDetectorAndQualityStrictPassRate": 0
    }
  ],
  "historicalControls": [
    {
      "id": "printing-press",
      "historicalZeroGpt": 19,
      "currentZeroGpt": 19,
      "historicalAiDetector": 28,
      "currentAiDetector": 73
    },
    {
      "id": "daily-planning",
      "historicalZeroGpt": 44.2,
      "currentZeroGpt": 44.2,
      "historicalAiDetector": 20,
      "currentAiDetector": 22
    },
    {
      "id": "sleep",
      "historicalZeroGpt": 27,
      "currentZeroGpt": 27,
      "historicalAiDetector": 9,
      "currentAiDetector": 19
    },
    {
      "id": "customer-support",
      "historicalZeroGpt": 49.5,
      "currentZeroGpt": 49.5,
      "historicalAiDetector": 19,
      "currentAiDetector": 3
    },
    {
      "id": "feedback",
      "historicalZeroGpt": 0,
      "currentZeroGpt": 0,
      "historicalAiDetector": 19,
      "currentAiDetector": 39
    },
    {
      "id": "cloud-storage",
      "historicalZeroGpt": 31.4,
      "currentZeroGpt": 31.4,
      "historicalAiDetector": 28,
      "currentAiDetector": 15
    }
  ],
  "conclusion": "The selected candidate was materially more reliable on the quality gate, but neither candidate achieved a strict pass across both detectors on the frozen holdout. The study does not support a guaranteed undetectable or detector-bypass claim.",
  "limitations": [
    "The holdout contained eight topics and two generations per candidate per topic.",
    "Only two browser-accessible detectors were used.",
    "A separate evaluation model, rather than a human panel, scored the factual-quality gate.",
    "Generation is stochastic and detector services can change after the experiment date.",
    "Average detector scores are descriptive and should not be interpreted as stable pass probabilities."
  ]
}
