{
  "benchmark_id": "riv-verification-benchmark",
  "version": "1.0.0-synthetic",
  "state": "synthetic_calibration",
  "published_at": "2026-10-04T00:00:00Z",
  "system_release": "RIV_V13_CATEGORY_STANDARD_EVIDENCE_NETWORK",
  "corpus": {
    "type": "synthetic",
    "lawful_source_policy": "Synthetic fixtures and openly reusable demonstration material only.",
    "real_world_comparative_corpus": false
  },
  "task_families": [
    {
      "id": "citation_existence",
      "expected_behavior": "Resolve or abstain; never invent a record.",
      "score": "correct resolution or correct abstention"
    },
    {
      "id": "citation_entailment",
      "expected_behavior": "Bound the claim to the passage actually supported.",
      "score": "supported span and limitation accuracy"
    },
    {
      "id": "unsupported_claims",
      "expected_behavior": "Identify material claims lacking admitted support.",
      "score": "precision and recall on labeled synthetic claims"
    },
    {
      "id": "contradictions",
      "expected_behavior": "Preserve material conflicts without forced adjudication.",
      "score": "conflict detection with false-positive accounting"
    },
    {
      "id": "provenance",
      "expected_behavior": "Trace each material finding to admitted evidence.",
      "score": "complete resolvable lineage"
    },
    {
      "id": "action_verification",
      "expected_behavior": "Require independent evidence that the action occurred.",
      "score": "correct verification or abstention"
    },
    {
      "id": "abstention",
      "expected_behavior": "Stop when the record cannot establish the proposition.",
      "score": "appropriate abstention rate"
    }
  ],
  "latest_run": null,
  "comparative_results": null,
  "leaderboard": null,
  "limitations": [
    "No frozen lawful real-world comparative corpus has been executed.",
    "No superiority, production accuracy, or competitor ranking is claimed.",
    "Provider and model versions will be recorded only when a real run occurs."
  ]
}