{
  "model": "tropa-2",
  "evaluationDate": "2026-08-31",
  "humanTextEvaluation": {
    "source": "FineWeb-2, crawls before 2022",
    "method": "Document scoring with sentence chunks and character-weighted aggregation",
    "aiGeneratedThreshold": 90,
    "likelyAiThreshold": 70,
    "languages": [
      { "code": "de", "name": "German", "nativeName": "Deutsch", "samples": 15881, "flaggedAt90": 4, "flaggedAt70": 10 },
      { "code": "fr", "name": "French", "nativeName": "Français", "samples": 15958, "flaggedAt90": 6, "flaggedAt70": 35 },
      { "code": "es", "name": "Spanish", "nativeName": "Español", "samples": 15986, "flaggedAt90": 7, "flaggedAt70": 45 },
      { "code": "it", "name": "Italian", "nativeName": "Italiano", "samples": 16071, "flaggedAt90": 1, "flaggedAt70": 19 },
      { "code": "pt", "name": "Portuguese", "nativeName": "Português", "samples": 16041, "flaggedAt90": 7, "flaggedAt70": 36 },
      { "code": "nl", "name": "Dutch", "nativeName": "Nederlands", "samples": 15900, "flaggedAt90": 0, "flaggedAt70": 24 }
    ]
  },
  "germanAiEvaluation": {
    "samples": 300,
    "generators": ["GPT-4o", "GPT-5.5"],
    "flaggedAt90": 286,
    "flaggedAt70": 300
  },
  "notes": [
    "Evaluation conducted by WasItAIGenerated, not an independent audit.",
    "Pre-2022 crawl dates are used as a proxy for human authorship. These multilingual reference samples are web texts, not a student-essay benchmark.",
    "The language table measures false positives on human reference texts. It does not measure AI recall for French, Spanish, Italian, Portuguese, or Dutch.",
    "The German AI sample contains raw generated text. It does not establish performance on humanized or translated German text.",
    "Observed zero false positives in a finite sample does not imply a zero population error rate.",
    "Score thresholds of 70 and 90 represent different operating points and must not be combined into one accuracy claim."
  ]
}
