{
  "name": "AI Music Detector Public Benchmark",
  "version": "v1.1",
  "updated": "2026-08-24",
  "engine_build": "historical-ensemble-1",
  "status": "engine_behaviour_measured_accuracy_pending",
  "license": "CC BY 4.0",
  "url": "https://aimusicdetector.com/benchmark",
  "engine_behaviour_study": {
    "id": "engine-behaviour-1",
    "title": "Engine behaviour study 1",
    "measuredOn": "2026-08-24",
    "engineBuild": "historical-ensemble-1",
    "tracks": 12,
    "conditions": 7,
    "detections": 168,
    "corpus": "12 reference tracks (20–23 s, 44.1 kHz stereo) generated procedurally by the site's own sample generator across 12 genre presets. No third-party or copyrighted audio is used.",
    "ladder": "lossless WAV · MP3 320 · 192 · 128 · 64 kbps · mono downmix · 10 s excerpt",
    "findings": [
      {
        "id": "determinism",
        "metric": "Repeat determinism",
        "value": "0.0 pp",
        "detail": "Each of the 84 files was analysed twice in the same research run. All 84 pairs returned an identical probability, so that historical build introduced no observed run-to-run variation."
      },
      {
        "id": "encode-stability",
        "metric": "Encode stability (mean |Δp| vs lossless)",
        "value": "2.4 pp",
        "detail": "Across the four MP3 bitrates, the mean shift was 1.8 pp at 320 kbps, 1.8 pp at 192 kbps, 2.3 pp at 128 kbps, and 3.5 pp at 64 kbps. The largest single shift observed was 26 pp, at 64 kbps."
      },
      {
        "id": "mono",
        "metric": "Mono downmix sensitivity",
        "value": "1.2 pp",
        "detail": "Collapsing stereo to mono moved the score by 1.2 pp on average, with a worst case of 11 pp, even though stereo-field features become unavailable once the channels are combined."
      },
      {
        "id": "excerpt",
        "metric": "Short-excerpt sensitivity",
        "value": "4.4 pp",
        "detail": "A 10 s excerpt of the same master moved the score by 4.4 pp on average, and by as much as 31 pp. This demonstrates excerpt sensitivity in that historical build; it is not a rule for the current provider."
      },
      {
        "id": "abstention",
        "metric": "Abstention behaviour",
        "value": "83 of 84 files inconclusive",
        "detail": "On single-render synthetic reference audio, the historical build declined to issue a verdict in almost every case. That build was configured to abstain rather than guess, and it is a reminder that an inconclusive result is a normal outcome."
      },
      {
        "id": "synthetic-risk",
        "metric": "Synthesiser-only false-positive risk",
        "value": "3 of 12 tracks read above 70%",
        "detail": "Three of the twelve renders — all synthesiser-only, quantized, single-take material — produced historical scores above 70%. This small procedural set demonstrates a possible confound, and this study reports that risk rather than hiding it."
      },
      {
        "id": "runtime",
        "metric": "Analysis time",
        "value": "406 ms median",
        "detail": "The median was 406 ms and the 90th percentile was 532 ms per track, as measured on the harness, excluding upload time. These harness timings do not describe the current third-party public service."
      }
    ],
    "caveats": [
      "This is a behavior and robustness study, not an accuracy study. It contains no AI-generated tracks and no verified human commercial recordings, so it produces no detection rate and no false-positive rate.",
      "The corpus is generated procedurally, which makes it reproducible but not representative of released music.",
      "All figures apply only to the retired historical-ensemble-1 research build. They do not evaluate the current public provider."
    ]
  },
  "engine_facts": [
    {
      "id": "models",
      "label": "Models in the ensemble",
      "value": "4"
    },
    {
      "id": "weights",
      "label": "Vote weights",
      "value": "spectral-artefact 2.2 · embedding 1.6 · acoustic 1.0 · extended-features 0.9"
    },
    {
      "id": "output-cap",
      "label": "Reported probability range",
      "value": "15–85%"
    },
    {
      "id": "shrinkage",
      "label": "Shrinkage toward 50%",
      "value": "0.94"
    },
    {
      "id": "confidence-bands",
      "label": "Confidence bands",
      "value": "moderate ≥ 0.50 · high ≥ 0.75"
    },
    {
      "id": "disagreement",
      "label": "Disagreement penalty",
      "value": "threshold 0.20 · penalty 1.1×"
    },
    {
      "id": "processing",
      "label": "Relationship to public detector",
      "value": "Separate historical research build"
    }
  ],
  "metrics": [
    {
      "id": "tpr",
      "name": "Detection rate (TPR)",
      "definition": "Share of known AI-generated tracks assigned an AI category by the evaluated classifier, counted per generator rather than pooled across generators."
    },
    {
      "id": "fpr",
      "name": "False-positive rate (FPR)",
      "definition": "Share of verified human-produced tracks assigned an AI category. We treat this as the primary metric, because a false accusation costs a musician more than a missed detection does."
    },
    {
      "id": "inconclusive",
      "name": "Inconclusive rate",
      "definition": "Share of tracks assigned an inconclusive outcome. Abstentions are reported on their own, not folded into the accuracy figures."
    },
    {
      "id": "calibration",
      "name": "Calibration error",
      "definition": "Calibration measure defined for any future classifier output that can support it; it will not be calculated by treating component indicators as an overall probability."
    },
    {
      "id": "stability",
      "name": "Encode stability",
      "definition": "Mean absolute change in a comparable numeric output between the lossless reading of a master and the same master rendered at 320, 192, 128 and 64 kbps."
    }
  ],
  "corpus": {
    "planned_tracks": 690,
    "generator_arms": [
      {
        "id": "suno",
        "generator": "Suno",
        "version": "v4 / v4.5",
        "planned": 60,
        "tpr": null,
        "inconclusive": null
      },
      {
        "id": "udio",
        "generator": "Udio",
        "version": "v1.5",
        "planned": 60,
        "tpr": null,
        "inconclusive": null
      },
      {
        "id": "stable-audio",
        "generator": "Stable Audio",
        "version": "2.0",
        "planned": 40,
        "tpr": null,
        "inconclusive": null
      },
      {
        "id": "elevenlabs",
        "generator": "ElevenLabs Music",
        "version": "current",
        "planned": 40,
        "tpr": null,
        "inconclusive": null
      },
      {
        "id": "riffusion",
        "generator": "Riffusion",
        "version": "FUZZ",
        "planned": 40,
        "tpr": null,
        "inconclusive": null
      },
      {
        "id": "seed-music",
        "generator": "Seed Music",
        "version": "current",
        "planned": 30,
        "tpr": null,
        "inconclusive": null
      },
      {
        "id": "minimax",
        "generator": "MiniMax Music",
        "version": "current",
        "planned": 30,
        "tpr": null,
        "inconclusive": null
      },
      {
        "id": "mureka",
        "generator": "Mureka",
        "version": "current",
        "planned": 30,
        "tpr": null,
        "inconclusive": null
      },
      {
        "id": "held-out",
        "generator": "Held-out generator",
        "version": "undisclosed until publication",
        "planned": 40,
        "tpr": null,
        "inconclusive": null
      }
    ],
    "control_arms": [
      {
        "id": "studio",
        "source": "Commercially released studio recordings",
        "planned": 80,
        "rationale": "Heavily processed human music is the most common false-positive scenario.",
        "fpr": null
      },
      {
        "id": "bedroom",
        "source": "Independent bedroom productions",
        "planned": 80,
        "rationale": "In-the-box production with stock plugins looks superficially synthetic.",
        "fpr": null
      },
      {
        "id": "acoustic",
        "source": "Live acoustic and single-room recordings",
        "planned": 60,
        "rationale": "The easiest control case; a failure here would be disqualifying.",
        "fpr": null
      },
      {
        "id": "electronic",
        "source": "Fully synthetic human-authored electronic music",
        "planned": 60,
        "rationale": "Quantised, synthesiser-only human work is the hardest control case.",
        "fpr": null
      },
      {
        "id": "hybrid",
        "source": "Hybrid human/AI tracks (AI stems in a human arrangement)",
        "planned": 40,
        "rationale": "Reported separately; there is no correct binary answer for these.",
        "fpr": null
      }
    ]
  },
  "limitations": [
    {
      "id": "lim-compression",
      "title": "Lossy compression is the dominant error source",
      "body": "Lossy encoding changes the submitted audio. Future results will be reported per source condition rather than pooled across bitrates."
    },
    {
      "id": "lim-drift",
      "title": "Generator versions move faster than benchmarks",
      "body": "A figure measured against one model release says little about the next release. Any future number will be stamped with the generator version and evaluated provider version."
    },
    {
      "id": "lim-hybrid",
      "title": "Hybrid tracks have no ground truth",
      "body": "A human arrangement built around a generated stem fits neither class. These tracks are reported as a separate arm and excluded from the headline figures."
    },
    {
      "id": "lim-adversarial",
      "title": "The benchmark is not adversarial",
      "body": "Deliberate evasion — re-recording, heavy remastering, aggressive time and pitch manipulation — belongs to a separate study. These figures should not be read as robustness against an attacker."
    },
    {
      "id": "lim-scope",
      "title": "No claim about individual tracks",
      "body": "A population-level rate does not carry over to a single file. No figure on this page should be used as evidence about a specific song or person."
    }
  ],
  "note": "Null result fields are unmeasured; they are never populated with estimates."
}