{
  "slug": "ai-detectors-accuracy-humanize-writing-2026",
  "title": "Are AI detectors accurate? Evidence and false-positive calculator",
  "description": "Check AI detector accuracy claims against published research. Explore false positives, base rates and what a detector flag can actually tell you.",
  "published": "2026-05-30",
  "updated": "2026-08-28",
  "sources": [
    {
      "id": "openai-2023",
      "title": "OpenAI: retired text classifier",
      "url": "https://openai.com/index/new-ai-classifier-for-indicating-ai-written-text/",
      "date": "2023-01-31",
      "kind": "Provider evaluation; historical",
      "population": "English challenge set; retired product",
      "finding": "Reported 26% detection of AI text and 9% false positives on human text. The classifier was withdrawn on 20 July 2023.",
      "limit": "Not an estimate of present-day detectors or a current product recommendation."
    },
    {
      "id": "stanford-2023",
      "title": "Stanford HAI: non-native English writing",
      "url": "https://hai.stanford.edu/news/ai-detectors-biased-against-non-native-english-writers",
      "date": "2023-05-15",
      "kind": "University report of research",
      "population": "91 TOEFL essays; seven detectors",
      "finding": "The reported average false-positive rate for the TOEFL essays was 61.22% across the tested detectors.",
      "limit": "A specific historical sample, not the rate for every non-native writer or every detector."
    },
    {
      "id": "raid-2024",
      "title": "RAID: shared detector benchmark (ACL 2024)",
      "url": "https://aclanthology.org/2024.acl-long.674/",
      "date": "2024-08",
      "kind": "Peer-reviewed conference paper",
      "population": "2024 paper: 6M+ generations, 11 generators, eight domains; 12 detectors evaluated",
      "finding": "Performance varied with unfamiliar generators, decoding choices and adversarial changes.",
      "limit": "Use the paper's dataset edition; the current repository contains a larger, expanded dataset."
    },
    {
      "id": "park-2026",
      "title": "Park, Jeong and Kim: editing style as a confound",
      "url": "https://arxiv.org/abs/2608.26710v1",
      "date": "2026-08-27",
      "kind": "arXiv v1; venue status not independently verified",
      "population": "135,389 original/edited document pairs from one academic editing service, 2018-2025; 13 detectors",
      "finding": "Editing changed detector scores in different directions. Baseline false-positive analysis used the pre-ChatGPT subset.",
      "limit": "Do not treat all 135,389 pairs as the baseline false-positive denominator, or this sample as a current commercial-detector ranking."
    }
  ],
  "faq": [
    [
      "Are AI detectors 99% accurate?",
      "There is no universal accuracy figure. A claim needs the detector version, threshold, dataset and test date. Accuracy also depends on the proportion of AI and human texts; a 99% claim is not a 99% probability that a flagged document is AI-written."
    ],
    [
      "Can a human-written essay be flagged as AI?",
      "Yes. That is a false positive. The size of the risk depends on the detector and population. Keep drafts, sources and revision history; a detector score alone does not establish authorship."
    ],
    [
      "Does this calculator check my writing?",
      "No. It calculates expected counts from rates you enter. It does not read text, run a detector or decide whether someone used AI."
    ],
    [
      "Are these ToolGlance benchmark results?",
      "No. The evidence table attributes published findings to their original researchers. The calculator and scenario CSV are ToolGlance arithmetic, not a new detector benchmark."
    ],
    [
      "Does a lower score after editing prove a text is human?",
      "No. A changed score shows that the detector responded differently. Edit for correctness and readability, follow the applicable disclosure policy, and do not use a score as proof of authorship."
    ]
  ],
  "correction": "28 August 2026: replaced an unsourced blanket 83-94% accuracy range with source-specific findings and an explicit base-rate calculation. The URL is unchanged."
}
