{
  "contentIdentitySha256": "20433480a0a0a56c3ffc61283c429f9b4bfb7430895388b80633c7675731f6ef",
  "coreTaskClasses": [
    "completion",
    "deployment",
    "generation",
    "package-migration",
    "performance-repair",
    "policy-minimization",
    "refactor",
    "repair",
    "replay-investigation"
  ],
  "kind": "genesis/genesisbench-benchmark-card-v0.1",
  "name": "GenesisBench",
  "nonClaims": [
    "Public cases are not held-out or temporal-clean.",
    "Benchmark score is not a general intelligence measure.",
    "Cost and latency do not improve substantive rank.",
    "A deterministic mock is conformance evidence only.",
    "Unknown training provenance is not clean provenance."
  ],
  "primaryMetrics": [
    "verified-pass-at-one",
    "bounded-pass-at-k",
    "conditional-quality-among-valid-solves"
  ],
  "protocolIdentitySha256": "bfff3ace2cddfe545cd090bf27471ae126f2c6c02d09da0fe67f6f3adbabacac",
  "purpose": "Measure how reliably an agent learns and engineers GenesisCode under frozen authority and bounded resources.",
  "requiredDisclosures": [
    "adapter",
    "all-attempts",
    "contamination",
    "cost",
    "hardware",
    "model-revision",
    "runtime",
    "scaffold"
  ],
  "safetyMetrics": [
    "authority-excess",
    "capability-requests",
    "invalid-attempts",
    "policy-scope",
    "resource-excess"
  ],
  "status": "project-controlled-preview",
  "unitOfIndependence": "task-lineage",
  "version": "0.1.0"
}
