{
 "schema_version": "0.3",
 "synthetic": false,
 "phase": "phase-0",
 "labels": [
  "phase-0",
  "operator-run",
  "operator-keyed",
  "unblinded",
  "operator-affiliated"
 ],
 "banner": "Phase 0: operator self-assessment. Questions, answer keys and scoring are run by Asymmetric Intelligence, which operates the systems scored. Results are not yet independently validated or maintained. They measure the fleet's progress against the TRACE method, not independent quality. Challenges are open now through our contact page; the checks become re-runnable by anyone when the repository opens.",
 "not_yet_in_place": [
  "No sealed split (all items are public dev items the fleet has seen)",
  "No external domain reviewers",
  "No independent keyholders",
  "No external entrants",
  "Judge\u2013human agreement (\u03ba) not independently measured"
 ],
 "run": {
  "id": "run-1",
  "window_start": "2026-10-06",
  "window_end": "2026-10-06",
  "repo_commit": "83fcd7c (+ branch tr2/bind; consumer snapshot commits in runs/run-1/NOTES.md)",
  "itemset_sha256": "n/a: no Track A items in run-1",
  "prereg_tag": "none: run-1 was not pre-registered",
  "harness_version": "trackb-adapter (tr2/bind, --gate check_source_provenance.py blob fb17d37a)",
  "judges": [],
  "kappa": {
   "status": "not_yet_measured",
   "reason": "No human-scored sample yet; no Track A judging in run-1."
  },
  "rerun_command": "see runs/run-1/NOTES.md section 'Reproduce' (adapter.py per consumer, then runs/run-1/build_scorecard.py)",
  "previous_run": "run-0"
 },
 "alpha_gate_progress": {
  "external_reviewers": [
   0,
   3
  ],
  "keyholders": [
   0,
   3
  ],
  "external_entrants_sealed": [
   0,
   3
  ],
  "kappa_target": 0.7
 },
 "consumers": [
  {
   "id": "advennt",
   "name": "Advennt",
   "url": "https://advennt.io",
   "union_enrolled": true,
   "labels": [
    "phase-0",
    "operator-run",
    "operator-keyed",
    "unblinded",
    "operator-affiliated"
   ],
   "track_b": {
    "records": 170,
    "citations": 2272,
    "checks": {
     "B1": {
      "status": "measured",
      "checked": 2272,
      "failing": 510,
      "by_type": {
       "blocked": 241,
       "unretrieved": 255,
       "throttled": 14
      },
      "fatal": true,
      "fatal_count": 145,
      "warning_count": 365,
      "failing_records_url": "runs/run-1/out/trackb-advennt-failing.jsonl"
     },
     "B2": {
      "status": "measured",
      "checked": 2272,
      "failing": 21,
      "by_type": {
       "provenance_mismatch": 21
      },
      "fatal": false,
      "fatal_count": 21,
      "warning_count": 0,
      "failing_records_url": "runs/run-1/out/trackb-advennt-failing.jsonl"
     },
     "B3": {
      "status": "measured",
      "checked": 740,
      "failing": 73,
      "by_type": {
       "t1_off_allowlist": 73
      },
      "fatal": true,
      "failing_records_url": "runs/run-1/out/trackb-advennt-failing.jsonl"
     },
     "B4": {
      "status": "pending_first_run"
     },
     "B5": {
      "status": "pending_first_run"
     },
     "B6": {
      "status": "measured",
      "checked": 170,
      "failing": 0,
      "by_type": {},
      "fatal": false,
      "failing_records_url": "runs/run-1/out/trackb-advennt-failing.jsonl"
     }
    },
    "b7_sampled_truth": {
     "status": "not_yet_measured",
     "reason": "B7 needs dev answer keys from P0-c; out of scope for P0-b."
    }
   },
   "track_a": {
    "split": "dev",
    "mode": "A3",
    "n_items": null,
    "metrics": {
     "accuracy": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "hallucination_rate": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "citation_existence": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "citation_recall": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "citation_precision": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "tier_precision": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "tier_recall": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "depth": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "breadth_correct": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "breadth_honest_gap": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "temporal_accuracy": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "correct_abstention": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "over_abstention": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "calibration_ece": {
      "status": "not_measurable",
      "reason": "No stated per-claim confidence until synthesised confidence is removed and real confidence is emitted (fleet rec P-2). Never imputed."
     }
    }
   },
   "top_gaps": []
  },
  {
   "id": "crypto",
   "name": "Crypto",
   "url": "https://cryptoassets.gi",
   "union_enrolled": true,
   "labels": [
    "phase-0",
    "operator-run",
    "operator-keyed",
    "unblinded",
    "operator-affiliated"
   ],
   "track_b": {
    "records": 170,
    "citations": 2498,
    "checks": {
     "B1": {
      "status": "measured",
      "checked": 2498,
      "failing": 867,
      "by_type": {
       "blocked": 650,
       "unretrieved": 204,
       "throttled": 13
      },
      "fatal": true,
      "fatal_count": 27,
      "warning_count": 840,
      "failing_records_url": "runs/run-1/out/trackb-crypto-failing.jsonl"
     },
     "B2": {
      "status": "measured",
      "checked": 1610,
      "failing": 6,
      "by_type": {
       "provenance_mismatch": 6
      },
      "fatal": true,
      "fatal_count": 6,
      "warning_count": 0,
      "failing_records_url": "runs/run-1/out/trackb-crypto-failing.jsonl"
     },
     "B3": {
      "status": "measured",
      "checked": 249,
      "failing": 21,
      "by_type": {
       "t1_off_allowlist": 21
      },
      "fatal": true,
      "failing_records_url": "runs/run-1/out/trackb-crypto-failing.jsonl"
     },
     "B4": {
      "status": "pending_first_run"
     },
     "B5": {
      "status": "pending_first_run"
     },
     "B6": {
      "status": "measured",
      "checked": 170,
      "failing": 0,
      "by_type": {},
      "fatal": false,
      "failing_records_url": "runs/run-1/out/trackb-crypto-failing.jsonl"
     }
    },
    "b7_sampled_truth": {
     "status": "not_yet_measured",
     "reason": "B7 needs dev answer keys from P0-c; out of scope for P0-b."
    }
   },
   "track_a": {
    "split": "dev",
    "mode": "A3",
    "n_items": null,
    "metrics": {
     "accuracy": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "hallucination_rate": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "citation_existence": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "citation_recall": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "citation_precision": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "tier_precision": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "tier_recall": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "depth": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "breadth_correct": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "breadth_honest_gap": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "temporal_accuracy": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "correct_abstention": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "over_abstention": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "calibration_ece": {
      "status": "not_measurable",
      "reason": "No stated per-claim confidence until synthesised confidence is removed and real confidence is emitted (fleet rec P-2). Never imputed."
     }
    }
   },
   "top_gaps": []
  },
  {
   "id": "data-protection",
   "name": "Data Protection",
   "url": "https://dataprotection.gi",
   "union_enrolled": true,
   "labels": [
    "phase-0",
    "operator-run",
    "operator-keyed",
    "unblinded",
    "operator-affiliated"
   ],
   "track_b": {
    "records": 170,
    "citations": 4004,
    "checks": {
     "B1": {
      "status": "measured",
      "checked": 4004,
      "failing": 784,
      "by_type": {
       "unretrieved": 644,
       "blocked": 140
      },
      "fatal": true,
      "fatal_count": 146,
      "warning_count": 638,
      "failing_records_url": "runs/run-1/out/trackb-data-protection-failing.jsonl"
     },
     "B2": {
      "status": "measured",
      "checked": 2980,
      "failing": 5,
      "by_type": {
       "provenance_mismatch": 5
      },
      "fatal": true,
      "fatal_count": 5,
      "warning_count": 0,
      "failing_records_url": "runs/run-1/out/trackb-data-protection-failing.jsonl"
     },
     "B3": {
      "status": "measured",
      "checked": 300,
      "failing": 17,
      "by_type": {
       "t1_off_allowlist": 17
      },
      "fatal": true,
      "failing_records_url": "runs/run-1/out/trackb-data-protection-failing.jsonl"
     },
     "B4": {
      "status": "pending_first_run"
     },
     "B5": {
      "status": "pending_first_run"
     },
     "B6": {
      "status": "measured",
      "checked": 170,
      "failing": 0,
      "by_type": {},
      "fatal": false,
      "failing_records_url": "runs/run-1/out/trackb-data-protection-failing.jsonl"
     }
    },
    "b7_sampled_truth": {
     "status": "not_yet_measured",
     "reason": "B7 needs dev answer keys from P0-c; out of scope for P0-b."
    }
   },
   "track_a": {
    "split": "dev",
    "mode": "A3",
    "n_items": null,
    "metrics": {
     "accuracy": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "hallucination_rate": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "citation_existence": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "citation_recall": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "citation_precision": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "tier_precision": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "tier_recall": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "depth": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "breadth_correct": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "breadth_honest_gap": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "temporal_accuracy": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "correct_abstention": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "over_abstention": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "calibration_ece": {
      "status": "not_measurable",
      "reason": "No stated per-claim confidence until synthesised confidence is removed and real confidence is emitted (fleet rec P-2). Never imputed."
     }
    }
   },
   "top_gaps": []
  },
  {
   "id": "financial-integrity",
   "name": "Financial Integrity",
   "url": "https://sentinel.gi",
   "union_enrolled": true,
   "labels": [
    "phase-0",
    "operator-run",
    "operator-keyed",
    "unblinded",
    "operator-affiliated"
   ],
   "track_b": {
    "records": 170,
    "citations": 21913,
    "checks": {
     "B1": {
      "status": "measured",
      "checked": 21913,
      "failing": 5636,
      "by_type": {
       "blocked": 5374,
       "unretrieved": 259,
       "throttled": 3
      },
      "fatal": true,
      "fatal_count": 874,
      "warning_count": 4762,
      "failing_records_url": "runs/run-1/out/trackb-financial-integrity-failing.jsonl"
     },
     "B2": {
      "status": "measured",
      "checked": 14428,
      "failing": 42,
      "by_type": {
       "provenance_mismatch": 42
      },
      "fatal": true,
      "fatal_count": 42,
      "warning_count": 0,
      "failing_records_url": "runs/run-1/out/trackb-financial-integrity-failing.jsonl"
     },
     "B3": {
      "status": "measured",
      "checked": 3641,
      "failing": 100,
      "by_type": {
       "t1_off_allowlist": 100
      },
      "fatal": true,
      "failing_records_url": "runs/run-1/out/trackb-financial-integrity-failing.jsonl"
     },
     "B4": {
      "status": "pending_first_run"
     },
     "B5": {
      "status": "pending_first_run"
     },
     "B6": {
      "status": "measured",
      "checked": 170,
      "failing": 0,
      "by_type": {},
      "fatal": false,
      "failing_records_url": "runs/run-1/out/trackb-financial-integrity-failing.jsonl"
     }
    },
    "b7_sampled_truth": {
     "status": "not_yet_measured",
     "reason": "B7 needs dev answer keys from P0-c; out of scope for P0-b."
    }
   },
   "track_a": {
    "split": "dev",
    "mode": "A3",
    "n_items": null,
    "metrics": {
     "accuracy": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "hallucination_rate": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "citation_existence": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "citation_recall": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "citation_precision": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "tier_precision": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "tier_recall": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "depth": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "breadth_correct": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "breadth_honest_gap": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "temporal_accuracy": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "correct_abstention": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "over_abstention": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "calibration_ece": {
      "status": "not_measurable",
      "reason": "No stated per-claim confidence until synthesised confidence is removed and real confidence is emitted (fleet rec P-2). Never imputed."
     }
    }
   },
   "top_gaps": []
  },
  {
   "id": "world-payments",
   "name": "World Payments Monitor",
   "url": "https://payments.gi",
   "union_enrolled": true,
   "labels": [
    "phase-0",
    "operator-run",
    "operator-keyed",
    "unblinded",
    "operator-affiliated"
   ],
   "track_b": {
    "records": 170,
    "citations": 30487,
    "checks": {
     "B1": {
      "status": "measured",
      "checked": 30487,
      "failing": 6105,
      "by_type": {
       "blocked": 3041,
       "unretrieved": 2931,
       "throttled": 133
      },
      "fatal": true,
      "fatal_count": 1266,
      "warning_count": 4839,
      "failing_records_url": "runs/run-1/out/trackb-world-payments-failing.jsonl"
     },
     "B2": {
      "status": "measured",
      "checked": 21645,
      "failing": 178,
      "by_type": {
       "provenance_mismatch": 178
      },
      "fatal": true,
      "fatal_count": 178,
      "warning_count": 0,
      "failing_records_url": "runs/run-1/out/trackb-world-payments-failing.jsonl"
     },
     "B3": {
      "status": "measured",
      "checked": 7561,
      "failing": 279,
      "by_type": {
       "t1_off_allowlist": 279
      },
      "fatal": true,
      "failing_records_url": "runs/run-1/out/trackb-world-payments-failing.jsonl"
     },
     "B4": {
      "status": "pending_first_run"
     },
     "B5": {
      "status": "pending_first_run"
     },
     "B6": {
      "status": "measured",
      "checked": 170,
      "failing": 0,
      "by_type": {},
      "fatal": false,
      "failing_records_url": "runs/run-1/out/trackb-world-payments-failing.jsonl"
     }
    },
    "b7_sampled_truth": {
     "status": "not_yet_measured",
     "reason": "B7 needs dev answer keys from P0-c; out of scope for P0-b."
    }
   },
   "track_a": {
    "split": "dev",
    "mode": "A3",
    "n_items": null,
    "metrics": {
     "accuracy": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "hallucination_rate": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "citation_existence": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "citation_recall": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "citation_precision": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "tier_precision": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "tier_recall": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "depth": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "breadth_correct": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "breadth_honest_gap": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "temporal_accuracy": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "correct_abstention": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "over_abstention": {
      "status": "not_yet_measured",
      "reason": "Track A runner, judges and dev items are not built yet (P0-c, P0-d); run-1 measures Track B only."
     },
     "calibration_ece": {
      "status": "not_measurable",
      "reason": "No stated per-claim confidence until synthesised confidence is removed and real confidence is emitted (fleet rec P-2). Never imputed."
     }
    }
   },
   "top_gaps": []
  }
 ],
 "baselines": [],
 "changelog": [
  "run-0: page scaffold; no results yet.",
  "run-1: first Track B run on recorded provenance of all five union consumers (published JID records). Results as found."
 ],
 "open_challenges": 0,
 "rulings_url": "https://github.com/<org>/trace-benchmark/blob/main/RULINGS.md"
}
