{
  "schema_version": 2,
  "manifest_ref": "multimodal-models:audio-checkpoint-benchmark",
  "benchmark": "ESC-10 subset of ESC-50",
  "evaluation_tier": "l2_public_benchmark_subset",
  "seeds": [
    42
  ],
  "evaluated_examples": 30,
  "prediction_source": "public_checkpoint",
  "metadata": {
    "dataset_source": "https://github.com/karolpiczak/ESC-50",
    "annotations_sha256": "1fff8678d80205d212bd386cb6614a1df71f4700d4be651aeacf2f3f06238509",
    "subset_protocol": "first three official ESC-10 examples per class in ESC-50 metadata order",
    "model_id": "laion/clap-htsat-unfused",
    "model_revision": "8fa0f1c6d0433df6e97c127f64b2a1d6c0dcda8a",
    "checkpoint_committed": false,
    "prompt_template": "This is a sound of {label}.",
    "cache_replay_verified": true,
    "runtime": {
      "platform": "Linux x86_64",
      "resolved_device": "cuda",
      "accelerator": "NVIDIA A30"
    }
  },
  "aggregate_metrics": {
    "zero_shot_top1_accuracy": 1.0,
    "zero_shot_top5_accuracy": 1.0,
    "examples": 30,
    "classes": 10
  },
  "formal_comparison": false,
  "claim_policy": "deterministic zero-shot checkpoint regression on a fixed 30-example ESC-10 subset; not the full supervised five-fold ESC-50 protocol",
  "evaluation_protocol": {
    "tier": "l2_public_benchmark_subset",
    "seeds": [
      42
    ],
    "formal_comparison": false,
    "claim_policy": "deterministic zero-shot checkpoint regression on a fixed 30-example ESC-10 subset; not the full supervised five-fold ESC-50 protocol"
  },
  "provenance": {
    "created_at": "2026-08-20T09:28:08.773812+00:00",
    "artifact_path": "docs/multimodal-models/metrics/esc10-clap-a30-30.json",
    "dataset_fingerprint": "sha256:1fff8678d80205d212bd386cb6614a1df71f4700d4be651aeacf2f3f06238509",
    "checkpoint_path": "local snapshot (not committed)"
  }
}
