{
  "title": "TextSight false-positive rate on human-written academic English (2026)",
  "publisher": "Lacewing Technologies (textsight.ai)",
  "licence": "CC BY 4.0",
  "n_documents": 1180,
  "n_by_group": {
    "esl": 759,
    "native": 421
  },
  "n_by_country": {
    "China": 42,
    "Japan": 85,
    "South Korea": 101,
    "Spain": 80,
    "Iran": 145,
    "United Kingdom": 102,
    "Italy": 71,
    "United States": 106,
    "Turkey": 82,
    "Australia": 51,
    "Ireland": 81,
    "Brazil": 153,
    "New Zealand": 81
  },
  "scoring_errors_excluded": 0,
  "source": "PubMed Central Open Access subset, CC BY / CC0 only",
  "publication_year_filter": 2018,
  "provenance_argument": "Every document was published in 2018. The GPT-3 API opened mid-2020 and ChatGPT launched 2022-11-30, so authorship pre-dates the models. Every AI flag on this corpus is therefore a false positive by construction.",
  "group_assignment": "First-author affiliation country, used as a PROXY for author first language. Papers carrying any affiliation from the opposing language group were excluded, so a US co-author cannot place a paper in the second-language arm and vice versa.",
  "text_preparation": "Abstracts only. Journal structural labels (Abstract, Background, Methods, Results and similar) stripped identically from both arms, because they were unevenly distributed (15.7% of native vs 4.3% of second-language documents opened with one) and are typesetting rather than authored prose. Length band 150-400 words. Non-Latin-script and results-table abstracts dropped.",
  "detector": {
    "endpoint": "/internal/analyze, mode=full",
    "decision_rule": "verdict.label == 'AI-Generated', which is the production decision. server/services/detection.ts only selects the wording after the model has decided.",
    "determinism": "Verified before the run: identical text scored 3 times returned identical values (sd = 0)."
  },
  "known_limitations": [
    "Affiliation country is a proxy for first language, not a measurement of it.",
    "Academic abstracts are frequently copy-edited by journals or paid editing services, which moves second-language prose toward native norms and therefore biases the measured gap DOWNWARD.",
    "Academic abstract register only. Does not generalise to student essays, blog posts or social media.",
    "Single publication year (2018) and a single detector version.",
    "The detector is English-only and this study concerns English-surface text only."
  ],
  "files": {
    "textsight-fpr-2026-per-document.csv": {
      "sha256": "e21dc3ab12f3b31f7ab7b1bbdbe5836478d47e3e2577693d6039d91daf645b2c",
      "bytes": 217716
    },
    "corpus.jsonl": {
      "sha256": "15ea8bbbfbd9c7e2c4902b4b04e280eda8571662c0cdfa12f9f276ffeb3b669d",
      "bytes": 2147601
    }
  }
}