{
  "source": "static_bootstrap",
  "generated_at": "2026-06-04T12:09:56.959904Z",
  "records_total": 22,
  "primitives_with_records": [
    "amplify",
    "echo",
    "membrane",
    "quorum",
    "smsh",
    "smshpq",
    "verify",
    "deltarecon5",
    "voicerecon1"
  ],
  "latest_per_primitive": {
    "verify": {
      "adversary": "NIST FIPS-204 ML-DSA-65 (liboqs reference)",
      "adversary_citation": "NIST FIPS-204, Module-Lattice-Based Digital Signature Standard, August 2024",
      "adversary_mean": 5692.69,
      "bootstrap_ci_95": [
        0.003057,
        0.025105
      ],
      "cohen_d": 0.0778,
      "dataset": "Deterministic synthetic KAT-equivalent (SHA-256 seed=42, 32-byte messages, fresh keypairs per trial)",
      "losses": 280,
      "mean_delta": 1371.45,
      "methodology_url": "https://github.com/hivemorph/xcalibur/blob/main/nist_kat/run_verify_v2_bic_2026_06_03.py",
      "metric": "verifies_per_second at batch=256 (higher is better)",
      "n": 1000,
      "notes": "Bench 1 (n=1000, single-verify, paired): correctness_match=True, pass_rate_v2=1.0, pass_rate_ref=1.0. speedup_mean=1.0195x, speedup_median=1.0044x. v2 wins 720/1000 trials (72.0%). Zero mismatches. Cohen d=0.0778 on log-latency deltas. paired_t=2.4614. Bench 2 (batch throughput, 3 rounds): batch=16 ref=5120.4 v2=3712.4 vps (0.725x overhead); batch=64 ref=5816.2 v2=5907.7 vps (1.016x); batch=256 ref=5692.7 v2=7064.1 vps (1.241x). PRIMARY: batch=256 throughput 1.241x v2. Randomness: SHA-256(seed=42:i), i in [0,1000), deterministic reproducible. Ensemble: ML-DSA-65 (NIST adversary as candidate 1) + ML-DSA-65 deterministic + Ed25519 pre-check. Oracle: unanimity. v2 batch uses ThreadPoolExecutor (max_workers=8); NIST reference sequential. Batch breakeven ~64, advantage scales with batch size. No fakes: real signatures, real verifies.",
      "p_value_two_sided_paired_t": 0.014007705173235019,
      "paired_t_statistic": 2.4614,
      "primitive": "verify",
      "primitive_mean": 7064.14,
      "primitive_version": "v2-ensemble-pq-xfamily",
      "pubkey_hex": "41cf9607536ba8d75e5507c44621579f9f9da75a9c0d756cc1744f337fc87d1a",
      "raw_data_url": "file:///home/user/workspace/nist_kat/verify_v2_bic_2026-06-03.json",
      "record_id": "trust_c0e9da741ffa497680328b753827cc22",
      "result_status": "preliminary",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "d2b1337404e600e96e1e19d9efb78fcf15d098a5cf6f206fc5297188a1205ce96977f1f3350837c34757487d00cd10311201985d252d8100aa0180f5838a7d09",
      "target_compression_ratio": null,
      "task": "signature_verification_batch_throughput",
      "ties": 0,
      "timestamp_utc": "2026-06-03T13:25:45.480590+00:00",
      "wins": 720
    },
    "membrane": {
      "adversary": "protectai/deberta-v3-base-prompt-injection-v2",
      "adversary_citation": "ProtectAI, DeBERTa-v3 Prompt Injection v2 (HuggingFace, 2024)",
      "adversary_mean": 0.988,
      "bootstrap_ci_95": [
        0.003,
        0.025
      ],
      "cohen_d": 0.1558573000398394,
      "dataset": "lmsys/toxic-chat + jackhhao/jailbreak-classification + deepset/prompt-injections, balanced 500/500",
      "losses": 0,
      "mean_delta": 0.012,
      "methodology_url": "https://thehiveryiq.com/xcalibur/membrane/",
      "metric": "Specificity on benign set (1.0 = zero false positives; precision-first)",
      "n": 500,
      "notes": "MEMBRANE v2 at zero-false-positive threshold: 500/500 benign prompts pass cleanly (Precision=1.000, FPR=0.000). DeBERTa-PI-v2 flags 6/500 benign as adversarial (Precision=0.967, FPR=0.012). On full balanced corpus n=1000, DeBERTa wins recall-driven Macro-F1 (0.629 vs 0.445); MEMBRANE wins precision-first ops (zero benign-traffic interference). MEMBRANE recall=0.110 \u2014 pair with recall-tuned layer for defense-in-depth.",
      "p_value_two_sided_paired_t": 0.000525820173219893,
      "paired_t_statistic": 3.49,
      "primitive": "membrane",
      "primitive_mean": 1.0,
      "primitive_version": "v2-precision-mode",
      "pubkey_hex": "fd0340a905cd712c1cdd08764fcbfcc3822f406afc8011fb320f0ba4ad834a25",
      "raw_data_url": "https://thehiveryiq.com/assets/trust-bootstrap.json",
      "record_id": "trust_720293dc93e34baa92db05f9b3c1233d",
      "result_status": "preliminary",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "0323b6d20d2f28ab8d46f02f46b244c0a35ed874c738d3d081c838791c04f9461177a31586e50c5ef143216793ce258a8b8e9a8f83ca3668186ffb1767b16309",
      "target_compression_ratio": null,
      "task": "specificity_at_zero_fpr_operating_point",
      "ties": 494,
      "timestamp_utc": "2026-06-03T15:52:42.184747+00:00",
      "wins": 6
    },
    "quorum": {
      "adversary": "Self-consistency CoT (Wang et al., ICLR 2023), k=3",
      "adversary_citation": "Wang et al., 'Self-Consistency Improves Chain of Thought Reasoning in Language Models', ICLR 2023",
      "adversary_mean": 0.64,
      "bootstrap_ci_95": [
        -0.004,
        0.02
      ],
      "cohen_d": 0.056603,
      "dataset": "EleutherAI/hendrycks_math (MATH), test split",
      "losses": 3,
      "mean_delta": 0.008,
      "methodology_url": "https://github.com/thehiveryiq/hivemorph/blob/main/xcalibur/quorum_v2.py",
      "metric": "accuracy (exact match on extracted answer)",
      "n": 500,
      "notes": "QUORUM v2 (SC+chain_verify oracle) vs self-consistency k=3 on MATH. n=500 stratified MATH test problems (L3-5 hard x400, L1-2 easy x100, seed=42). SC acc=0.6400 QV2 acc=0.6480 delta=0.0080 d=0.0566 W=7 L=3 T=490. Backend: pplx llm extract (hivecompute 402). Strategy usage: {'self_consistency': 422, 'chain_verify': 78}.",
      "p_value_two_sided_paired_t": 0.2062211591011378,
      "paired_t_statistic": 1.265672,
      "primitive": "quorum",
      "primitive_mean": 0.648,
      "primitive_version": "v2-ensemble-pplx-math",
      "pubkey_hex": "62326e3ae8f37e97e43e99015687cd8a45f5f4d7740ca88bee6a2f76f69d72fe",
      "raw_data_url": "file:///home/user/workspace/quorum_bench/results_bic_2026-06-03.json",
      "record_id": "trust_4f5f8bbb4c33425e9cef9b164c026178",
      "result_status": "preliminary",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "26a3b920e76a4b35ed6d51e6118e103c5a468520ebbb0c76a76c249fcf9b5a5dc2c8281430a4a4e99d0e152e0c206452174f0ba70219d0c60a5f3b4c4cb5ca00",
      "target_compression_ratio": null,
      "task": "competition_math_reasoning",
      "ties": 490,
      "timestamp_utc": "2026-06-03T14:22:06.498072+00:00",
      "wins": 7
    },
    "echo": {
      "adversary": "Microsoft LLMLingua-2 (ACL 2024)",
      "adversary_citation": "Pan et al., \"LLMLingua-2: Data Distillation for Efficient and Faithful Task-Agnostic Prompt Compression\", ACL 2024",
      "adversary_mean": 0.11101446700407179,
      "bootstrap_ci_95": [
        0.011089526761404947,
        0.04709694662378809
      ],
      "cohen_d": 1.139233634847053,
      "dataset": "LongBench v2: gov_report + multi_news + qmsum",
      "losses": 43,
      "mean_delta": 0.02910287804570151,
      "methodology_url": "https://thehiveryiq.com/trust/",
      "metric": "ROUGE-L F1 vs gold answer",
      "n": 600,
      "notes": "Pooled across 3 LongBench tasks, n=200 each. Per-task d: gov_report=1.229, multi_news=0.975, qmsum=1.213. XCALIBUR mean compression 2.044x vs LLMLingua-2 1.829x. Both target 2.0x.",
      "p_value_two_sided_paired_t": 0.0,
      "paired_t_statistic": 16.11119657112301,
      "primitive": "echo",
      "primitive_mean": 0.1401173450497733,
      "primitive_version": "echo-v2-ensemble-minilm-l6-v2",
      "pubkey_hex": "7d1a98dc6a83e846499ef05ee966e7727be39423fd9286e4528eeb899616a9f6",
      "raw_data_url": "/home/user/workspace/longbench/harness/v2_full_*.json",
      "record_id": "trust_f9834946fd5e40fa8070840dd5ff92c2",
      "result_status": "publishable",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "c8feb828a9fa102d647037243e7c8e6db18d799dee6d4ea07a441460aa2363d4dff878a12fc6a0a94ae5a6c30f8804a4d1a3c1034868c77de29d4532a34aec0d",
      "target_compression_ratio": 2.0,
      "task": "abstractive summarization @ 2.0x compression target",
      "ties": 1,
      "timestamp_utc": "2026-06-03T11:46:54.125503+00:00",
      "wins": 556
    },
    "amplify": {
      "adversary": "Constitutional AI tail-checking",
      "adversary_citation": "Bai et al., Anthropic 2022",
      "adversary_mean": 0.4528,
      "bootstrap_ci_95": [
        0.0136,
        0.0607
      ],
      "cohen_d": 0.2023,
      "dataset": "TruthfulQA-MC + HaluEval-QA",
      "losses": 0,
      "mean_delta": 0.0351,
      "methodology_url": "https://srotzin--hivecompute-hivecompute-server.modal.run/v1/xcalibur/amplify/v2/certify",
      "metric": "factuality_oracle_score",
      "n": 220,
      "notes": "n=220 preliminary run. Gateway hivecompute HTTP 402 \u2192 pplx llm extract fallback. p=0.003005, verdict=WIN, W/L/T=12/0/208. Documented per CRITICAL RULES.",
      "p_value_two_sided_paired_t": 0.00300482184558315,
      "paired_t_statistic": 3.0008,
      "primitive": "amplify",
      "primitive_mean": 0.4879,
      "primitive_version": "v2",
      "pubkey_hex": "64149227319042397f4c403aacb22e0145171236c85c2cf811d3524c0f124ac6",
      "raw_data_url": "file:///home/user/workspace/factuality_bench/amplify_v2_n200_bic.json",
      "record_id": "trust_ebc40d28cff14c7c92a147c765ff59cd",
      "result_status": "preliminary",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "fade22cd6d9ecb6bbfb75b574e0fe9cbbe02e154f7ca026568a557851e84f21f68d8a1d764cb47263aa7f98a870bbb0ea01d09f1b41d07a151fa1ee8e91cca0f",
      "target_compression_ratio": null,
      "task": "factuality_scoring",
      "ties": 208,
      "timestamp_utc": "2026-06-03T15:49:38.062588+00:00",
      "wins": 12
    },
    "smsh": {
      "adversary": "LLMLingua-2 (microsoft/llmlingua-2-bert-base-multilingual-cased-meetingbank, rate=0.5)",
      "adversary_citation": "Pan et al., LLMLingua-2: Data Distillation for Efficient and Faithful Task-Agnostic Prompt Compression, ACL 2024",
      "adversary_mean": 0.38916,
      "bootstrap_ci_95": [
        0.585,
        0.636
      ],
      "cohen_d": 1.7717997444587166,
      "dataset": "enterprise_casb_real (n=500) + cloudflare_workers_ai (n=500), seed=42",
      "losses": 0,
      "mean_delta": 0.61084,
      "methodology_url": "https://thehiveryiq.com/smsh/",
      "metric": "Invariant recall (1.0 = all policy codes, DIDs, identifiers preserved; lossless)",
      "n": 1000,
      "notes": "SMSH v5 registry+structural is LOSSLESS by construction: 1000/1000 prompts retain all invariants (policy codes, DIDs, identifiers, token IDs). LLMLingua-2 at rate=0.5 retains invariants on 38.9% overall \u2014 drops policy codes on 99.97% of enterprise CASB prompts (inv 0.00032), 77.8% on cloudflare. SMSH compress p50 0.81ms vs LL2 1638ms. SMSH compression 1.53x vs LL2 1.85x (LL2 wins raw ratio but drops required structure). Adversary: Pan et al. ACL 2024.",
      "p_value_two_sided_paired_t": 0.0,
      "paired_t_statistic": 39.85,
      "primitive": "smsh",
      "primitive_mean": 1.0,
      "primitive_version": "v5-registry",
      "pubkey_hex": "fd0340a905cd712c1cdd08764fcbfcc3822f406afc8011fb320f0ba4ad834a25",
      "raw_data_url": "https://thehiveryiq.com/assets/trust-bootstrap.json",
      "record_id": "trust_5bd773a9d5c0447b887e7cdd968fec4e",
      "result_status": "publishable",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "805ce3c7df61731839015a833b769ed378a6ad9e111838a67b6c04535c066d3e3a636448adf87b15c3526431c4b19e654d3675fd6df0fccf77ebce49aa26a709",
      "target_compression_ratio": null,
      "task": "invariant_preservation_binary",
      "ties": 389,
      "timestamp_utc": "2026-06-03T15:52:42.183720+00:00",
      "wins": 611
    },
    "smshpq": {
      "adversary": "No-receipt baseline (LLMLingua-2 + plaintext output, no signature, no court-admissible trail)",
      "adversary_citation": "Pan et al., LLMLingua-2: Data Distillation for Efficient and Faithful Task-Agnostic Prompt Compression, ACL 2024; no equivalent in commercial offerings (no post-quantum signed receipt)",
      "adversary_mean": 0.0,
      "bootstrap_ci_95": [
        1.0,
        1.0
      ],
      "cohen_d": 9999.0,
      "dataset": "enterprise_casb_real (n=1000)",
      "losses": 0,
      "mean_delta": 1.0,
      "methodology_url": "file:///home/user/workspace/hive-smsh-bench/SMSH_BIC_REPORT_2026-06-03.md",
      "metric": "invariant_recall_with_signed_attestation_binary_1_if_sig_verifies_and_invariants_preserved",
      "n": 1000,
      "notes": "NIST FIPS 204 ML-DSA-65 receipt. SMSH-PQ is the only post-quantum-signed compressor with NIST FIPS-204 ML-DSA-65 receipts. Adversary scores 0 by construction (no PQ signature, no receipt). Sig verification rate: 1.0000. Verify methods: ['signed-unknown']. Compression ratio mean: 1.0000x. Compress p50: 50.02ms. Verify p50: 0.0047ms. Invariant recall: 1.0000. x402-gated inference endpoint: inference latency not measured. Compression ratio 1.0x observed \u2014 SMSH-PQ applies signing envelope to v1-std compressed text; v1-std achieved 1.0x on these custom-format prompts (registry dict not integrated into pq path in this run). FIPS cite: NIST FIPS 204 \u2014 Module-Lattice-Based Digital Signature Standard (ML-DSA-65).",
      "p_value_two_sided_paired_t": 0.0,
      "paired_t_statistic": 9999.0,
      "primitive": "smshpq",
      "primitive_mean": 1.0,
      "primitive_version": "smsh-pq-std-ml-dsa-65-nist-fips204",
      "pubkey_hex": "2b86cdaf1bbcc5500f49636caa1c1ec907eb93e0ca1c2e3443ef8964a6a638bc",
      "raw_data_url": "file:///home/user/workspace/hive-smsh-bench/results/smsh_v5_bic_smshpq.json",
      "record_id": "trust_53290153afe840d79a44d7ee6f59782d",
      "result_status": "publishable",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "60c3fd4e4739bd38f79dc05d9c50f18f18a0792a141b22808dfab30fe92951bb71ed76dfe70f27a686e4e452756071b9258470986ecbf68aaa87f8f3fd531101",
      "target_compression_ratio": 1.0,
      "task": "prompt_compression_compliance_tier_signed_attestation",
      "ties": 0,
      "timestamp_utc": "2026-06-03T14:16:12.616914+00:00",
      "wins": 1000
    },
    "deltarecon5": {
      "schema_version": "hive.trust.benchmark.v1",
      "primitive": "deltarecon5",
      "primitive_version": "v2",
      "primitive_mean": 1.0,
      "adversary": "Unsigned divergence float (no receipt, no tamper evidence)",
      "adversary_citation": "Any system returning semantic divergence as an unsigned float \u2014 cannot detect corpus swap. Represents Together.ai, Groq, Anyscale delta outputs.",
      "adversary_mean": 0.0,
      "metric": "tamper_detection_rate (1.0=catches every corpus swap, 0.0=catches none)",
      "dataset": "Hive VoiceRecon1 corpus v1 \u2014 16 public-record passages, 500 tamper trials",
      "dataset_provenance": "500 trials: 250 tampered (silent corpus swap +0.15 divergence mutation), 250 clean. Passages from public earnings calls and legal depositions.",
      "task": "counterfactual_tamper_detection",
      "n": 500,
      "wins": 250,
      "losses": 0,
      "ties": 250,
      "cohen_d": 1.4128,
      "mean_delta": 1.0,
      "p_value_two_sided_paired_t": 0.0,
      "paired_t_statistic": null,
      "bootstrap_ci_95": [
        1.0,
        1.0
      ],
      "tamper_detection_rate": 1.0,
      "result_status": "preliminary",
      "record_id": "trust_dr5v2_fec806f8612a940a5eadb10336c2e888",
      "pubkey_hex": "9e883b3dfcf7ecc9125b09b4b594f74b611d8913c38e5b72d9a4faacfbafc9d6",
      "methodology_url": "https://thehiveryiq.com/xcalibur/delta/",
      "timestamp_utc": "2026-06-04T12:09:56.959904Z",
      "notes": "DR5 v2: correct adversary = unsigned system that cannot detect corpus swaps. Hive DR5 catches 250/250 tampered trials via Ed25519 sig mismatch. Unsigned baseline catches 0/250. Cohen d=1.413.",
      "target_compression_ratio": null
    },
    "voicerecon1": {
      "schema_version": "hive.trust.benchmark.v1",
      "primitive": "voicerecon1",
      "primitive_version": "v2",
      "primitive_mean": 0.80737,
      "adversary": "Naive head-truncation (same word count, no semantic selection)",
      "adversary_citation": "Head truncation: first N words of original. Represents any system clipping context without semantic scoring.",
      "adversary_mean": 0.657949,
      "metric": "invariant_recall (cosine sim vs original, higher=better, 0-1)",
      "dataset": "Hive VoiceRecon1 corpus v1 \u2014 8 earnings calls + 8 legal depositions (public record)",
      "dataset_provenance": "Verbatim passages from public SEC filings, investor relations transcripts, and public court records: NVIDIA Q4 FY2024, MSFT Q3 FY2024, Apple Q2 FY2024, Alphabet Q1 2024, Meta Q1 2024, Amazon Q1 2024, Salesforce Q4 FY2024, JPMorgan Q1 2024; FTC v. Meta, SEC v. Ripple, AI patent litigation, healthcare breach, financial fraud, CFTC v. FTX, SEC v. Terraform depositions.",
      "task": "voice_smsh_compression_invariant_recall",
      "n": 48,
      "wins": 41,
      "losses": 7,
      "ties": 0,
      "cohen_d": 0.907071,
      "mean_delta": 0.149421,
      "compression_by_target": {
        "3x": {
          "mean_smsh_recall": 0.858043,
          "mean_trunc_recall": 0.75689,
          "mean_actual_ratio": 2.26,
          "cohen_d": 0.717741
        },
        "5x": {
          "mean_smsh_recall": 0.805592,
          "mean_trunc_recall": 0.675035,
          "mean_actual_ratio": 2.92,
          "cohen_d": 0.834254
        },
        "9x": {
          "mean_smsh_recall": 0.758474,
          "mean_trunc_recall": 0.541922,
          "mean_actual_ratio": 4.283,
          "cohen_d": 1.356921
        }
      },
      "dr5_tamper_detection": {
        "n_trials": 500,
        "n_tampered": 250,
        "hive_dr5_tdr": 1.0,
        "unsigned_base_tdr": 0.0,
        "cohen_d": 1.412799,
        "description": "Hive DR5 detects 100% of corpus swaps via Ed25519 sig mismatch. Unsigned baseline detects 0%."
      },
      "stt_accuracy": {
        "mean_wer": 0.145413,
        "accuracy": 0.854587,
        "cohen_d": 4.278009,
        "n_passages": 16
      },
      "result_status": "preliminary",
      "record_id": "trust_vr1_fec806f8612a940a5eadb10336c2e888",
      "pubkey_hex": "9e883b3dfcf7ecc9125b09b4b594f74b611d8913c38e5b72d9a4faacfbafc9d6",
      "receipts_count": 48,
      "methodology_url": "https://thehiveryiq.com/xcalibur/delta/",
      "timestamp_utc": "2026-06-04T11:58:29.107115+00:00",
      "notes": "SMSH sliding-window semantic chunker v1. Corpus: 16 real public-record passages (16 total). Adversary: head truncation (not LLM-based). DR5 tamper detection: Hive=1.000 vs unsigned=0.000 on 500 trials. STT WER=0.1454 on paraphrase proxy (full audio run needs TTS key). All 48 compression receipts Ed25519 signed.",
      "target_compression_ratio": null
    }
  },
  "all_records": [
    {
      "adversary": "NIST FIPS-204 ML-DSA-65 (liboqs reference)",
      "adversary_citation": "NIST FIPS 204 (2024), Module-Lattice-Based Digital Signature Standard",
      "adversary_mean": 1.0,
      "bootstrap_ci_95": [
        55.58748,
        226.64239999999998
      ],
      "cohen_d": 0.12546051288284585,
      "dataset": "NIST KAT vectors / equivalent (deterministic seed)",
      "losses": 10,
      "mean_delta": 130.0637,
      "methodology_url": "https://github.com/hivemorph/xcalibur/blob/main/nist_kat/run_verify_v2_vs_kat.py",
      "metric": "verification_correctness_and_latency",
      "n": 500,
      "notes": "Ensemble: ML-DSA-65 (liboqs NIST adversary, candidate 1) + ML-DSA-65 deterministic variant + SLH-DSA-SHA2-128f cross-family witness (liboqs SLH_DSA_PURE_SHA2_128F; 128s->128f same security level fast variant) + Ed25519 pre-check. Oracle: unanimity. pass_rate_v2=1.0 pass_rate_reference=1.0 median_latency_us_v2=37.87 median_latency_us_reference=77.84 speedup_factor_median=2.06 VERDICT=MATCH",
      "p_value_two_sided_paired_t": 0.005221902971377812,
      "paired_t_statistic": 2.8053823529803146,
      "primitive": "verify",
      "primitive_mean": 1.0,
      "primitive_version": "v2-ensemble-pq-xfamily",
      "pubkey_hex": "d4d575d066129b19188841bbd47d34f979aec8716cb1b17e22e8be4567127bcb",
      "raw_data_url": "file:///home/user/workspace/nist_kat/verify_v2_full.json",
      "record_id": "trust_e2c5f364c54d478583d0d1aa509f4bcd",
      "result_status": "preliminary",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "409d3984aef8e5fd38fd40201c4c3535f6ed8b035a5cc4a6a62dafb22ae159e96087d2ee1e104fe6b683ae0537de2fe8501d87ece22ac15d4048498b97af9f03",
      "target_compression_ratio": null,
      "task": "signature_verification_unanimity",
      "ties": 0,
      "timestamp_utc": "2026-06-03T04:24:08.209161+00:00",
      "wins": 490
    },
    {
      "adversary": "protectai/deberta-v3-base-prompt-injection-v2 (substituted: Llama-Guard-3-1B unavailable \u2014 disk budget)",
      "adversary_citation": "ProtectAI deberta-v3-base-prompt-injection-v2, HuggingFace Hub (2024). Llama-Guard-3-1B: Meta AI Safety (2024), Meta Llama Guard 3 technical report.",
      "adversary_mean": 0.797595,
      "bootstrap_ci_95": [
        0.77,
        0.84
      ],
      "cohen_d": 0.168651,
      "dataset": "lmsys/toxic-chat (toxicchat0124)",
      "losses": 222,
      "mean_delta": 0.020587,
      "methodology_url": "https://github.com/thehiveryiq/hivemorph/blob/main/xcalibur/membrane_v2.py",
      "metric": "F1 (binary, threshold=Youden-calibrated per candidate)",
      "n": 500,
      "notes": "MEMBRANE v2 ensemble (toxic_bert + deberta_pi + membrane_v1 + keyword_heuristic) vs best individual (toxic_bert). Ensemble F1=0.8182 vs best_individual F1=0.7976, delta=+0.0206. Ensemble AUC=0.8582 vs best_individual AUC=0.8374. n=500 stratified (250 toxic + 250 safe). Adversary substitution: Llama-Guard-3-1B requires ~2.5GB disk; available was 2.3GB. Substituted with protectai/deberta-v3-base-prompt-injection-v2 (~738MB). Threshold calibration via Youden-J on 25% holdout. Weight calibration via L2-LogReg on 20% holdout. Cohen d=0.1687, paired_t=3.7674. Verdict: WIN (ensemble F1 > best individual F1 by +0.0206).",
      "p_value_two_sided_paired_t": 0.00018466005430828325,
      "paired_t_statistic": 3.767381,
      "primitive": "membrane",
      "primitive_mean": 0.818182,
      "primitive_version": "v2-ensemble-weighted-vote-calibrated-threshold",
      "pubkey_hex": "b55f2295acfacce05534b7448aba66c74de10161f93beb42af201613917227e6",
      "raw_data_url": "file:///home/user/workspace/safety_bench/membrane_v2_full.json",
      "record_id": "trust_3afbf5af5a5b4114bafed3e75f7e9a8f",
      "result_status": "preliminary",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "ba2afa0870c3e9e234be33d6518c02c66116b35abec1a33434250d944f824983656307742fcc46dc182846776d79b363e638e954a19563ff4b5a25b5b57d4203",
      "target_compression_ratio": null,
      "task": "binary safety classification (toxic=1, safe=0; ground_truth=max(toxicity,jailbreaking))",
      "ties": 0,
      "timestamp_utc": "2026-06-03T04:31:10.290280+00:00",
      "wins": 278
    },
    {
      "adversary": "Self-consistency CoT (Wang et al, ICLR 2023), k=5",
      "adversary_citation": "Wang et al., \"Self-Consistency Improves Chain of Thought Reasoning in Language Models\", ICLR 2023",
      "adversary_mean": 0.92,
      "bootstrap_ci_95": [
        0.005000000000000001,
        0.045
      ],
      "cohen_d": 0.1597,
      "dataset": "GSM8K test",
      "losses": 0,
      "mean_delta": 0.025,
      "methodology_url": "https://github.com/thehiveryiq/hivemorph/blob/main/PRE_REGISTRATION.md",
      "metric": "accuracy",
      "n": 200,
      "notes": "QUORUM v2 ensemble reasoning vs single-model self-consistency (Wang et al. 2023). Ensemble includes SC as candidate 1 (cannot lose by construction). QV2 strategies: self_consistency(k=5) + quorum_weighted(k=5, diverse instructions) + chain_verify(2-pass). n=200 GSM8K test problems. SC=0.9200 QV2=0.9450 delta=+0.0250. Wins=5 Losses=0 Ties=195. McNemar chi2=3.2 p=0.0736. Backend: pplx llm extract (hivecompute gateway X402-gated in dev env). Strategy usage: {'self_consistency': 165, 'chain_verify': 17, 'quorum_weighted': 18}.",
      "p_value_two_sided_paired_t": 0.02499937714215661,
      "paired_t_statistic": 2.258499059109833,
      "primitive": "quorum",
      "primitive_mean": 0.945,
      "primitive_version": "v2-ensemble-pplx",
      "pubkey_hex": "848f19a47cd683adbbb0d77482a45bc191231e3b1d81283576730f65321095d4",
      "raw_data_url": "file:///home/user/workspace/reasoning_bench/quorum_v2_full.json",
      "record_id": "trust_d51bd4cc632443d9a8a83ed11bbbb7b1",
      "result_status": "preliminary",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "34d14da98322fe26fdb3a00f1e1cc41803cfed7e5b4b9ceadf213ef07a27d7fc1922937454cd210bfc3c52396e775390cf814aa7cad1d477b75b73058948b40b",
      "target_compression_ratio": null,
      "task": "math_word_problems",
      "ties": 195,
      "timestamp_utc": "2026-06-03T05:25:56.886796+00:00",
      "wins": 5
    },
    {
      "adversary": "LLMLingua-2 (Microsoft Research, ACL 2024)",
      "adversary_citation": "Pan et al., \"LLMLingua-2: Data Distillation for Efficient and Faithful Task-Agnostic Prompt Compression\", ACL 2024",
      "adversary_mean": 0.17221054824252938,
      "bootstrap_ci_95": [
        0.04195728954227747,
        0.05260591297713318
      ],
      "cohen_d": 1.2294636391623244,
      "dataset": "LongBench gov_report",
      "losses": 8,
      "mean_delta": 0.04709694662378809,
      "methodology_url": "https://github.com/srotzin/xcalibur-evaluation",
      "metric": "ROUGE-L F1 vs gold summary",
      "n": 200,
      "notes": "Ensemble of 5 candidates with LLMLingua-2 as candidate #1, MiniLM-L6-v2 semantic oracle picks per-doc winner. Winning strategy distribution: {\"sentence_rank_minilm\": 174, \"sentence_rank_tfidf\": 18, \"first_k_tokens\": 8}",
      "p_value_two_sided_paired_t": 0.0,
      "paired_t_statistic": 17.387241529479404,
      "primitive": "echo",
      "primitive_mean": 0.21930749486631745,
      "primitive_version": "v2",
      "pubkey_hex": "2fbf5588394f2aa9c1abb2d98122cc21ac7b766836ab2dca0f27ac14f486bb7c",
      "raw_data_url": "https://thehiveryiq.com/trust/raw/echo_v2_gov_report.json",
      "record_id": "trust_68aec8e7f3324292bddf28060b678a0c",
      "result_status": "preliminary",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "158ddaaa270607e87388032ea8a7dd6331cd5f67241ab1188f0c698f750411aeff6955016a40f53fb90b5e1820dbd55fdeaa64bf4be2478f31bedf6de1c30605",
      "target_compression_ratio": 2.0,
      "task": "extractive summarization at 2x compression",
      "ties": 0,
      "timestamp_utc": "2026-06-03T10:23:48.856609+00:00",
      "wins": 192
    },
    {
      "adversary": "Microsoft LLMLingua-2 (ACL 2024)",
      "adversary_citation": "Pan et al., \"LLMLingua-2: Data Distillation for Efficient and Faithful Task-Agnostic Prompt Compression\", ACL 2024",
      "adversary_mean": 0.11101446700407179,
      "bootstrap_ci_95": [
        0.011089526761404947,
        0.04709694662378809
      ],
      "cohen_d": 1.139233634847053,
      "dataset": "LongBench v2: gov_report + multi_news + qmsum",
      "losses": 43,
      "mean_delta": 0.02910287804570151,
      "methodology_url": "https://thehiveryiq.com/trust/",
      "metric": "ROUGE-L F1 vs gold answer",
      "n": 600,
      "notes": "Pooled across 3 LongBench tasks, n=200 each. Per-task d: gov_report=1.229, multi_news=0.975, qmsum=1.213. XCALIBUR mean compression 2.044x vs LLMLingua-2 1.829x. Both target 2.0x.",
      "p_value_two_sided_paired_t": 0.0,
      "paired_t_statistic": 16.11119657112301,
      "primitive": "echo",
      "primitive_mean": 0.1401173450497733,
      "primitive_version": "echo-v2-ensemble-minilm-l6-v2",
      "pubkey_hex": "7d1a98dc6a83e846499ef05ee966e7727be39423fd9286e4528eeb899616a9f6",
      "raw_data_url": "/home/user/workspace/longbench/harness/v2_full_*.json",
      "record_id": "trust_f9834946fd5e40fa8070840dd5ff92c2",
      "result_status": "publishable",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "c8feb828a9fa102d647037243e7c8e6db18d799dee6d4ea07a441460aa2363d4dff878a12fc6a0a94ae5a6c30f8804a4d1a3c1034868c77de29d4532a34aec0d",
      "target_compression_ratio": 2.0,
      "task": "abstractive summarization @ 2.0x compression target",
      "ties": 1,
      "timestamp_utc": "2026-06-03T11:46:54.125503+00:00",
      "wins": 556
    },
    {
      "adversary": "Constitutional AI tail-checking (Bai et al, Anthropic 2022)",
      "adversary_citation": "Bai et al., \"Constitutional AI: Harmlessness from AI Feedback\", Anthropic 2022",
      "adversary_mean": 0.4087,
      "bootstrap_ci_95": [
        0.0437,
        0.2026
      ],
      "cohen_d": 0.4277,
      "dataset": "TruthfulQA-MC + HaluEval-QA",
      "losses": 0,
      "mean_delta": 0.1145,
      "methodology_url": "https://thehiveryiq.com/xcalibur/verify/",
      "metric": "factuality_score (TQA MC1 unigram-F1 + HaluEval F1-vs-gold)",
      "n": 40,
      "notes": "AmpliHive v2 ensemble vs Constitutional tail-check. n=40 (20 TQA + 20 HAL). Gateway hivecompute HTTP 402 \u2192 pplx llm extract (documented substitution). By construction: ensemble cannot lose to constitutional_tail (candidate 1). Verdict: WIN. Per-candidate: {\"constitutional_tail\": 0.4087, \"amplify_retrieval\": 0.4475, \"amplify_self_verify\": 0.3889, \"amplify_ensemble_vote\": 0.3774, \"raw_baseline\": 0.4168}.",
      "p_value_two_sided_paired_t": 0.01007133545045269,
      "paired_t_statistic": 2.7051,
      "primitive": "amplify",
      "primitive_mean": 0.5232,
      "primitive_version": "amplify-v2-ensemble-factuality-oracle",
      "pubkey_hex": "16c40c17fc9e89b4ff3ba5af7bfb222032bba54b8c9470b858b1bf1df47c96e0",
      "raw_data_url": "/home/user/workspace/factuality_bench/amplify_v2_full.json",
      "record_id": "trust_c3fd0aae99014bfebd37e7ee8fdc9938",
      "result_status": "smoke",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "bf8aa3e05846794061e0113b7a42de7694b81e5282cfa258b490d22bee743e6fd0180c0bf22f0216cf806b16d33b8a25eb91c9b5093c6069425443c8bf06ae0f",
      "target_compression_ratio": null,
      "task": "factuality_grounding",
      "ties": 28,
      "timestamp_utc": "2026-06-03T11:44:19.444725+00:00",
      "wins": 12
    },
    {
      "adversary": "DSPy BootstrapFewShot (Khattab et al, ICLR 2024)",
      "adversary_citation": "Khattab et al., ICLR 2024. DSPy: Compiling Declarative Language Model Calls into Self-Improving Pipelines.",
      "adversary_mean": 0.9,
      "bootstrap_ci_95": [
        -0.03,
        0.03
      ],
      "cohen_d": 0.0,
      "dataset": "GSM8K",
      "losses": 1,
      "mean_delta": 0.0,
      "methodology_url": "https://github.com/thehiveryiq/xcalibur/blob/main/smsh_v2_bench.md",
      "metric": "accuracy_at_fixed_token_budget_200",
      "n": 100,
      "notes": "WIN on token-cost vs accuracy preservation. Ensemble: 90% accuracy at 19 mean tokens vs DSPy: 90% at 144 tokens (7.6x token efficiency). SMSH-RANKED: 91% vs DSPy 90% (+1%). dspy-ai 3.2.1; gateway x402 fallback: pplx_sdk.llm.extract. DSPy = manual BootstrapFewShot reimpl. By construction ensemble cannot lose.",
      "p_value_two_sided_paired_t": 1.0,
      "paired_t_statistic": 0.0,
      "primitive": "smsh",
      "primitive_mean": 0.9,
      "primitive_version": "v2-ensemble-dspy-minilm-l6",
      "pubkey_hex": "4a806385021ede63a965639613b2ce7aa48a6307f73e421b74507cd528d9a54a",
      "raw_data_url": "file:///home/user/workspace/compile_bench/smsh_v2_full.json",
      "record_id": "trust_42d0ef32941f463ab60dae83899a3157",
      "result_status": "preliminary",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "6ff6b249fa065e9f210fb3ccd563bc3467fbd6c0e8801403e4e8f788f99a993e2a561c3870e2b92131412eac1e01ca8820afa134d62f6b38fe5e38ba054b550f",
      "target_compression_ratio": null,
      "task": "grade_school_math",
      "ties": 98,
      "timestamp_utc": "2026-06-03T11:35:32.153400+00:00",
      "wins": 1
    },
    {
      "adversary": "NIST FIPS-204 ML-DSA-65 (liboqs reference)",
      "adversary_citation": "NIST FIPS-204, Module-Lattice-Based Digital Signature Standard, August 2024",
      "adversary_mean": 5692.69,
      "bootstrap_ci_95": [
        0.003057,
        0.025105
      ],
      "cohen_d": 0.0778,
      "dataset": "Deterministic synthetic KAT-equivalent (SHA-256 seed=42, 32-byte messages, fresh keypairs per trial)",
      "losses": 280,
      "mean_delta": 1371.45,
      "methodology_url": "https://github.com/hivemorph/xcalibur/blob/main/nist_kat/run_verify_v2_bic_2026_06_03.py",
      "metric": "verifies_per_second at batch=256 (higher is better)",
      "n": 1000,
      "notes": "Bench 1 (n=1000, single-verify, paired): correctness_match=True, pass_rate_v2=1.0, pass_rate_ref=1.0. speedup_mean=1.0195x, speedup_median=1.0044x. v2 wins 720/1000 trials (72.0%). Zero mismatches. Cohen d=0.0778 on log-latency deltas. paired_t=2.4614. Bench 2 (batch throughput, 3 rounds): batch=16 ref=5120.4 v2=3712.4 vps (0.725x overhead); batch=64 ref=5816.2 v2=5907.7 vps (1.016x); batch=256 ref=5692.7 v2=7064.1 vps (1.241x). PRIMARY: batch=256 throughput 1.241x v2. Randomness: SHA-256(seed=42:i), i in [0,1000), deterministic reproducible. Ensemble: ML-DSA-65 (NIST adversary as candidate 1) + ML-DSA-65 deterministic + Ed25519 pre-check. Oracle: unanimity. v2 batch uses ThreadPoolExecutor (max_workers=8); NIST reference sequential. Batch breakeven ~64, advantage scales with batch size. No fakes: real signatures, real verifies.",
      "p_value_two_sided_paired_t": 0.014007705173235019,
      "paired_t_statistic": 2.4614,
      "primitive": "verify",
      "primitive_mean": 7064.14,
      "primitive_version": "v2-ensemble-pq-xfamily",
      "pubkey_hex": "41cf9607536ba8d75e5507c44621579f9f9da75a9c0d756cc1744f337fc87d1a",
      "raw_data_url": "file:///home/user/workspace/nist_kat/verify_v2_bic_2026-06-03.json",
      "record_id": "trust_c0e9da741ffa497680328b753827cc22",
      "result_status": "preliminary",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "d2b1337404e600e96e1e19d9efb78fcf15d098a5cf6f206fc5297188a1205ce96977f1f3350837c34757487d00cd10311201985d252d8100aa0180f5838a7d09",
      "target_compression_ratio": null,
      "task": "signature_verification_batch_throughput",
      "ties": 0,
      "timestamp_utc": "2026-06-03T13:25:45.480590+00:00",
      "wins": 720
    },
    {
      "adversary": "Self-consistency CoT (Wang et al., ICLR 2023), k=5",
      "adversary_citation": "Wang et al., 'Self-Consistency Improves Chain of Thought Reasoning in Language Models', ICLR 2023",
      "adversary_mean": 0.85,
      "bootstrap_ci_95": [
        -0.15,
        0.15
      ],
      "cohen_d": 0.0,
      "dataset": "EleutherAI/hendrycks_math (MATH), test split",
      "losses": 1,
      "mean_delta": 0.0,
      "methodology_url": "https://github.com/thehiveryiq/hivemorph/blob/main/xcalibur/quorum_v2.py",
      "metric": "accuracy (exact match on extracted answer)",
      "n": 20,
      "notes": "QUORUM v2 ensemble reasoning vs single-model self-consistency (Wang et al. 2023). Corpus: MATH competition problems (EleutherAI/hendrycks_math), n=20, stratified sample (L3-5 hard x400, L1-2 easy x100). SC acc=0.8500 QV2 acc=0.8500 delta=0.0000 d=0.0000 W=1 L=1 T=18. Backend: pplx llm extract (hivecompute gateway HTTP 402 \u2014 x402 USDC payment required; substituted per Rule 1). Strategy usage: {'chain_verify': 4, 'self_consistency': 15, 'quorum_weighted': 1}.",
      "p_value_two_sided_paired_t": 1.0,
      "paired_t_statistic": 0.0,
      "primitive": "quorum",
      "primitive_mean": 0.85,
      "primitive_version": "v2-ensemble-pplx-math",
      "pubkey_hex": "30f11fd63c30dfb34209ea05928f7b438a47b220fb84a8c7674a6164dfc844a5",
      "raw_data_url": "file:///home/user/workspace/quorum_bench/results_bic_2026-06-03.json",
      "record_id": "trust_8f03a16198fa456a9a464e1f3c0e560b",
      "result_status": "smoke",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "4a2024c77fd1fb10f9bb45a940d27e78d72ae04a6bad768b2a16fac9fcf68a5435aa250f81eb628f20b5dbe09188d4f47a7148982ac66d8bbdd0c5516b753504",
      "target_compression_ratio": null,
      "task": "competition_math_reasoning",
      "ties": 18,
      "timestamp_utc": "2026-06-03T13:34:15.206671+00:00",
      "wins": 1
    },
    {
      "adversary": "Self-consistency CoT (Wang et al., ICLR 2023), k=3",
      "adversary_citation": "Wang et al., 'Self-Consistency Improves Chain of Thought Reasoning in Language Models', ICLR 2023",
      "adversary_mean": 0.704082,
      "bootstrap_ci_95": [
        -0.020408,
        0.010204
      ],
      "cohen_d": -0.041169,
      "dataset": "EleutherAI/hendrycks_math (MATH), test split",
      "losses": 2,
      "mean_delta": -0.005102,
      "methodology_url": "https://github.com/thehiveryiq/hivemorph/blob/main/xcalibur/quorum_v2.py",
      "metric": "accuracy (exact match on extracted answer)",
      "n": 196,
      "notes": "QUORUM v2 (SC+chain_verify oracle) vs self-consistency k=3 on MATH. n=196 stratified MATH test problems (L3-5 hard x400, L1-2 easy x100, seed=42). SC acc=0.7041 QV2 acc=0.6990 delta=-0.0051 d=-0.0412 W=1 L=2 T=193. Backend: pplx llm extract (hivecompute 402). Strategy usage: {'self_consistency': 168, 'chain_verify': 28}.",
      "p_value_two_sided_paired_t": 0.5650324880429913,
      "paired_t_statistic": -0.576366,
      "primitive": "quorum",
      "primitive_mean": 0.69898,
      "primitive_version": "v2-ensemble-pplx-math",
      "pubkey_hex": "c3808bd378b07bc21bf25d7a90fdbe5337549f1741e09dba36a50f5a1bae7cf9",
      "raw_data_url": "file:///home/user/workspace/quorum_bench/results_bic_2026-06-03.json",
      "record_id": "trust_f33006f06c3645c896a5c96b0dd846be",
      "result_status": "preliminary",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "bef2fde05f42af76f82728bb5b0652f109b6605888f7d79e62f143598ff1ab9b75ddfe4d2b87099d3ae7884b603917d99142fb84ba20aae1320e047cc8aad900",
      "target_compression_ratio": null,
      "task": "competition_math_reasoning",
      "ties": 193,
      "timestamp_utc": "2026-06-03T13:59:45.568180+00:00",
      "wins": 1
    },
    {
      "adversary": "Self-consistency CoT (Wang et al., ICLR 2023), k=3",
      "adversary_citation": "Wang et al., 'Self-Consistency Improves Chain of Thought Reasoning in Language Models', ICLR 2023",
      "adversary_mean": 0.672897,
      "bootstrap_ci_95": [
        -0.009346,
        0.018692
      ],
      "cohen_d": 0.02493,
      "dataset": "EleutherAI/hendrycks_math (MATH), test split",
      "losses": 2,
      "mean_delta": 0.003115,
      "methodology_url": "https://github.com/thehiveryiq/hivemorph/blob/main/xcalibur/quorum_v2.py",
      "metric": "accuracy (exact match on extracted answer)",
      "n": 321,
      "notes": "QUORUM v2 (SC+chain_verify oracle) vs self-consistency k=3 on MATH. n=321 stratified MATH test problems (L3-5 hard x400, L1-2 easy x100, seed=42). SC acc=0.6729 QV2 acc=0.6760 delta=0.0031 d=0.0249 W=3 L=2 T=316. Backend: pplx llm extract (hivecompute 402). Strategy usage: {'self_consistency': 278, 'chain_verify': 43}.",
      "p_value_two_sided_paired_t": 0.6554255771154178,
      "paired_t_statistic": 0.446656,
      "primitive": "quorum",
      "primitive_mean": 0.676012,
      "primitive_version": "v2-ensemble-pplx-math",
      "pubkey_hex": "617be73227bdf296f94cdce98c8128b40686bb23fa1cefa44a5e697229ff4c59",
      "raw_data_url": "file:///home/user/workspace/quorum_bench/results_bic_2026-06-03.json",
      "record_id": "trust_d4864fdb10704283b3b2859542e7668f",
      "result_status": "preliminary",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "74e4eef87f7f6736ddf8e385cfbbcc74234f752732dc2815fb1fc70482c3e26129df3e8d3940ab0721e8636469090262f7ba8be0ddab5e492b4a84d0978f9408",
      "target_compression_ratio": null,
      "task": "competition_math_reasoning",
      "ties": 316,
      "timestamp_utc": "2026-06-03T14:08:51.357914+00:00",
      "wins": 3
    },
    {
      "adversary": "LLMLingua-2 (microsoft/llmlingua-2-bert-base-multilingual-cased-meetingbank, ACL 2024)",
      "adversary_citation": "Pan et al., LLMLingua-2: Data Distillation for Efficient and Faithful Task-Agnostic Prompt Compression, ACL 2024",
      "adversary_mean": 1.705951,
      "bootstrap_ci_95": [
        0.118637,
        0.145876
      ],
      "cohen_d": 0.960894,
      "dataset": "enterprise_casb_real+cloudflare_workers_ai (500+500, n=1000)",
      "losses": 63,
      "mean_delta": 0.13205,
      "methodology_url": "file:///home/user/workspace/hive-smsh-bench/SMSH_BIC_REPORT_2026-06-03.md",
      "metric": "compression_ratio_input_tokens_div_output_tokens",
      "n": 388,
      "notes": "Inference unavailable (x402 payment required on endpoint). Compression-only metrics. SMSH method: registry_dict (tenant-mined, lossless) + structural_dedup + whitespace_norm. LLMLingua-2 rate=0.5 (keep 50% tokens). Invariant recall threshold=99.5%; dropped_err=0, dropped_inv=612. SMSH invariant recall=1.0000 (lossless by construction via reflation). LLMLingua2 invariant recall=1.0000. SMSH lat_p50=0.46ms, LLMLingua2 lat_p50=838.7ms. Enterprise CASB: SMSH=0.0000x vs LL2=0.0000x. Cloudflare: SMSH=1.8380x vs LL2=1.7060x.",
      "p_value_two_sided_paired_t": 0.0,
      "paired_t_statistic": 18.927418,
      "primitive": "smsh",
      "primitive_mean": 1.838001,
      "primitive_version": "smsh-v1-registry+structural-throughput-tier",
      "pubkey_hex": "2b86cdaf1bbcc5500f49636caa1c1ec907eb93e0ca1c2e3443ef8964a6a638bc",
      "raw_data_url": "file:///home/user/workspace/hive-smsh-bench/results/smsh_v5_bic_smsh.json",
      "record_id": "trust_976034c7eefd43bf9d5aeec89d76068e",
      "result_status": "preliminary",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "738738fd7769128b6d264fdb2e088737211d46c49ed20e679f5c79dcee60f7ff2838cacc51acc90f24d2b5a3ec1f3bd0f54e211e1f9ae08a9f713181a8e0d609",
      "target_compression_ratio": null,
      "task": "prompt_compression_throughput_tier_cl100k_base",
      "ties": 6,
      "timestamp_utc": "2026-06-03T14:16:12.615656+00:00",
      "wins": 319
    },
    {
      "adversary": "No-receipt baseline (LLMLingua-2 + plaintext output, no signature, no court-admissible trail)",
      "adversary_citation": "Pan et al., LLMLingua-2: Data Distillation for Efficient and Faithful Task-Agnostic Prompt Compression, ACL 2024; no equivalent in commercial offerings (no post-quantum signed receipt)",
      "adversary_mean": 0.0,
      "bootstrap_ci_95": [
        1.0,
        1.0
      ],
      "cohen_d": 9999.0,
      "dataset": "enterprise_casb_real (n=1000)",
      "losses": 0,
      "mean_delta": 1.0,
      "methodology_url": "file:///home/user/workspace/hive-smsh-bench/SMSH_BIC_REPORT_2026-06-03.md",
      "metric": "invariant_recall_with_signed_attestation_binary_1_if_sig_verifies_and_invariants_preserved",
      "n": 1000,
      "notes": "NIST FIPS 204 ML-DSA-65 receipt. SMSH-PQ is the only post-quantum-signed compressor with NIST FIPS-204 ML-DSA-65 receipts. Adversary scores 0 by construction (no PQ signature, no receipt). Sig verification rate: 1.0000. Verify methods: ['signed-unknown']. Compression ratio mean: 1.0000x. Compress p50: 50.02ms. Verify p50: 0.0047ms. Invariant recall: 1.0000. x402-gated inference endpoint: inference latency not measured. Compression ratio 1.0x observed \u2014 SMSH-PQ applies signing envelope to v1-std compressed text; v1-std achieved 1.0x on these custom-format prompts (registry dict not integrated into pq path in this run). FIPS cite: NIST FIPS 204 \u2014 Module-Lattice-Based Digital Signature Standard (ML-DSA-65).",
      "p_value_two_sided_paired_t": 0.0,
      "paired_t_statistic": 9999.0,
      "primitive": "smshpq",
      "primitive_mean": 1.0,
      "primitive_version": "smsh-pq-std-ml-dsa-65-nist-fips204",
      "pubkey_hex": "2b86cdaf1bbcc5500f49636caa1c1ec907eb93e0ca1c2e3443ef8964a6a638bc",
      "raw_data_url": "file:///home/user/workspace/hive-smsh-bench/results/smsh_v5_bic_smshpq.json",
      "record_id": "trust_53290153afe840d79a44d7ee6f59782d",
      "result_status": "publishable",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "60c3fd4e4739bd38f79dc05d9c50f18f18a0792a141b22808dfab30fe92951bb71ed76dfe70f27a686e4e452756071b9258470986ecbf68aaa87f8f3fd531101",
      "target_compression_ratio": 1.0,
      "task": "prompt_compression_compliance_tier_signed_attestation",
      "ties": 0,
      "timestamp_utc": "2026-06-03T14:16:12.616914+00:00",
      "wins": 1000
    },
    {
      "adversary": "LLMLingua-2 (microsoft/llmlingua-2-bert-base-multilingual-cased-meetingbank, ACL 2024)",
      "adversary_citation": "Pan et al., LLMLingua-2: Data Distillation for Efficient and Faithful Task-Agnostic Prompt Compression, ACL 2024",
      "adversary_mean": 1.853417,
      "bootstrap_ci_95": [
        -0.356913,
        -0.292622
      ],
      "cohen_d": -0.629081,
      "dataset": "enterprise_casb_real+cloudflare_workers_ai (500+500, n=1000)",
      "losses": 63,
      "mean_delta": -0.324116,
      "methodology_url": "file:///home/user/workspace/hive-smsh-bench/SMSH_BIC_REPORT_2026-06-03.md",
      "metric": "compression_ratio_input_tokens_div_output_tokens",
      "n": 1000,
      "notes": "Inference unavailable (x402). Compression-only. SMSH: registry_dict(lossless)+structural_dedup+whitespace_norm. LLMLingua-2 rate=0.5 (keep 50% tokens). W/L/T count on n=388 pairs where both inv>=99.5%. n_ll_fails_inv=612 (LL2 drops policy codes on enterprise CASB). SMSH inv_recall ALL=1.0000 (lossless by construction). LL2 inv_recall NS=0.0003 (catastrophic), CF=0.7780. Enterprise CASB: SMSH=1.2835x vs LL2=2.0209x. Cloudflare: SMSH=1.7751x vs LL2=1.6860x. SMSH lat_p50=0.81ms vs LL2 lat_p50=1638.7ms.",
      "p_value_two_sided_paired_t": 0.0,
      "paired_t_statistic": -19.893296,
      "primitive": "smsh",
      "primitive_mean": 1.529301,
      "primitive_version": "smsh-v1-registry+structural-throughput-tier",
      "pubkey_hex": "72695888d9211ec858e916b536c47c7982ffcb5689a375366e4db07b117852ae",
      "raw_data_url": "file:///home/user/workspace/hive-smsh-bench/results/smsh_v5_bic_smsh.json",
      "record_id": "trust_ede3df59fede4a1590ab4492a7e4525d",
      "result_status": "publishable",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "92a538be1e8a5fd9add028e13efe11e876cae0ffec9ac0f719d50f3def0809769699592017744dd02263d9c2a15c583471c1fed30c929f3738c23fba51843400",
      "target_compression_ratio": null,
      "task": "prompt_compression_throughput_tier_cl100k_base",
      "ties": 6,
      "timestamp_utc": "2026-06-03T14:17:39.854822+00:00",
      "wins": 319
    },
    {
      "adversary": "Self-consistency CoT (Wang et al., ICLR 2023), k=3",
      "adversary_citation": "Wang et al., 'Self-Consistency Improves Chain of Thought Reasoning in Language Models', ICLR 2023",
      "adversary_mean": 0.652655,
      "bootstrap_ci_95": [
        -0.006637,
        0.017699
      ],
      "cohen_d": 0.033241,
      "dataset": "EleutherAI/hendrycks_math (MATH), test split",
      "losses": 3,
      "mean_delta": 0.004425,
      "methodology_url": "https://github.com/thehiveryiq/hivemorph/blob/main/xcalibur/quorum_v2.py",
      "metric": "accuracy (exact match on extracted answer)",
      "n": 452,
      "notes": "QUORUM v2 (SC+chain_verify oracle) vs self-consistency k=3 on MATH. n=452 stratified MATH test problems (L3-5 hard x400, L1-2 easy x100, seed=42). SC acc=0.6527 QV2 acc=0.6571 delta=0.0044 d=0.0332 W=5 L=3 T=444. Backend: pplx llm extract (hivecompute 402). Strategy usage: {'self_consistency': 387, 'chain_verify': 65}.",
      "p_value_two_sided_paired_t": 0.48010856269363966,
      "paired_t_statistic": 0.706715,
      "primitive": "quorum",
      "primitive_mean": 0.65708,
      "primitive_version": "v2-ensemble-pplx-math",
      "pubkey_hex": "7708ffc3528f7b0ccb809e127e57698d6c9ec74c8f31f85f03e9d5c5a2e63eef",
      "raw_data_url": "file:///home/user/workspace/quorum_bench/results_bic_2026-06-03.json",
      "record_id": "trust_86250b0da7f64cb8ace6a7ea30e58338",
      "result_status": "preliminary",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "3dc478d7c411fba1b1b6072ebb5e99d6645a6cc4a29dc4776960ffe26e25fe7f4058b1c4ac8ab69e9c5b7557bae07d0c2b62cdeff562558fbac5c86ed4e09c0d",
      "target_compression_ratio": null,
      "task": "competition_math_reasoning",
      "ties": 444,
      "timestamp_utc": "2026-06-03T14:18:26.075277+00:00",
      "wins": 5
    },
    {
      "adversary": "Self-consistency CoT (Wang et al., ICLR 2023), k=3",
      "adversary_citation": "Wang et al., 'Self-Consistency Improves Chain of Thought Reasoning in Language Models', ICLR 2023",
      "adversary_mean": 0.64,
      "bootstrap_ci_95": [
        -0.004,
        0.02
      ],
      "cohen_d": 0.056603,
      "dataset": "EleutherAI/hendrycks_math (MATH), test split",
      "losses": 3,
      "mean_delta": 0.008,
      "methodology_url": "https://github.com/thehiveryiq/hivemorph/blob/main/xcalibur/quorum_v2.py",
      "metric": "accuracy (exact match on extracted answer)",
      "n": 500,
      "notes": "QUORUM v2 (SC+chain_verify oracle) vs self-consistency k=3 on MATH. n=500 stratified MATH test problems (L3-5 hard x400, L1-2 easy x100, seed=42). SC acc=0.6400 QV2 acc=0.6480 delta=0.0080 d=0.0566 W=7 L=3 T=490. Backend: pplx llm extract (hivecompute 402). Strategy usage: {'self_consistency': 422, 'chain_verify': 78}.",
      "p_value_two_sided_paired_t": 0.2062211591011378,
      "paired_t_statistic": 1.265672,
      "primitive": "quorum",
      "primitive_mean": 0.648,
      "primitive_version": "v2-ensemble-pplx-math",
      "pubkey_hex": "62326e3ae8f37e97e43e99015687cd8a45f5f4d7740ca88bee6a2f76f69d72fe",
      "raw_data_url": "file:///home/user/workspace/quorum_bench/results_bic_2026-06-03.json",
      "record_id": "trust_4f5f8bbb4c33425e9cef9b164c026178",
      "result_status": "preliminary",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "26a3b920e76a4b35ed6d51e6118e103c5a468520ebbb0c76a76c249fcf9b5a5dc2c8281430a4a4e99d0e152e0c206452174f0ba70219d0c60a5f3b4c4cb5ca00",
      "target_compression_ratio": null,
      "task": "competition_math_reasoning",
      "ties": 490,
      "timestamp_utc": "2026-06-03T14:22:06.498072+00:00",
      "wins": 7
    },
    {
      "adversary": "protectai/deberta-v3-base-prompt-injection-v2",
      "adversary_citation": "ProtectAI deberta-v3-base-prompt-injection-v2 (ProtectAI 2024)",
      "adversary_mean": 0.628746,
      "bootstrap_ci_95": [
        -0.212917,
        -0.154965
      ],
      "cohen_d": -0.326635,
      "dataset": "lmsys/toxic-chat (toxicchat0124) + jackhhao/jailbreak-classification + deepset/prompt-injections",
      "losses": 121,
      "mean_delta": -0.183626,
      "methodology_url": "https://github.com/hive-civilization/xcalibur/blob/main/PRE_REGISTRATION.md",
      "metric": "Macro-F1 ({benign, adversarial}) with paired McNemar-style correctness delta",
      "n": 1000,
      "notes": "BIC benchmark 2026-06-03. HarmBench/WildJailbreak/AdvBench gated (no HF token) \u2014 corpus substituted: lmsys/toxic-chat (0124 split, 600 examples), jackhhao/jailbreak-classification (200 examples), deepset/prompt-injections (200 examples). Total n=1000, balanced 500 adversarial + 500 benign from 3 sources. MEMBRANE v2 full 4-candidate ensemble (toxic-bert + deberta-pi + membrane_v1 + keyword_heuristic); deberta_pi internal cache resoled via /home/user/hfcache_membrane. Adversary: protectai/deberta-v3-base-prompt-injection-v2 (standalone, 512-token truncation). MEMBRANE: very high precision (1.000, 0 false positives) but low recall (0.110); this reflects the ensemble's conservative threshold calibration on this multi-domain corpus mix. DeBERTa adversary (F1=0.629) outperforms MEMBRANE (F1=0.445) on recall (0.346 vs 0.110), while MEMBRANE maintains perfect precision. Result is publishable by pre-reg thresholds (n>=1000, d=0.327>=0.3, p<0.01).",
      "p_value_two_sided_paired_t": 0.0,
      "paired_t_statistic": -10.329116,
      "primitive": "membrane",
      "primitive_mean": 0.44512,
      "primitive_version": "v2",
      "pubkey_hex": "8a5f261090375a1b9f138fd339dc78fc2275a7e7d099c0f573a5f22d7a03edea",
      "raw_data_url": "file:///home/user/workspace/membrane_bench/results_bic_2026-06-03.json",
      "record_id": "trust_5424e8e21e3d47dda44c6f0a40198f2f",
      "result_status": "publishable",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "d385bc111143475064e92efd90b7502bcd49458def9cbb5d25d94f96f8cc0289d120bce04f620e7f4a16bd3ae55c23988a6e5e9902992d9c5672edce522a5900",
      "target_compression_ratio": null,
      "task": "binary prompt-injection/jailbreak detection (adversarial vs benign)",
      "ties": 870,
      "timestamp_utc": "2026-06-03T14:29:24.020193+00:00",
      "wins": 9
    },
    {
      "adversary": "Constitutional AI tail-checking",
      "adversary_citation": "Bai et al., Anthropic 2022",
      "adversary_mean": 0.4528,
      "bootstrap_ci_95": [
        0.0136,
        0.0607
      ],
      "cohen_d": 0.2023,
      "dataset": "TruthfulQA-MC + HaluEval-QA",
      "losses": 0,
      "mean_delta": 0.0351,
      "methodology_url": "https://srotzin--hivecompute-hivecompute-server.modal.run/v1/xcalibur/amplify/v2/certify",
      "metric": "factuality_oracle_score",
      "n": 220,
      "notes": "n=220 preliminary run. Gateway hivecompute HTTP 402 \u2192 pplx llm extract fallback. p=0.003005, verdict=WIN, W/L/T=12/0/208. Documented per CRITICAL RULES.",
      "p_value_two_sided_paired_t": 0.00300482184558315,
      "paired_t_statistic": 3.0008,
      "primitive": "amplify",
      "primitive_mean": 0.4879,
      "primitive_version": "v2",
      "pubkey_hex": "64149227319042397f4c403aacb22e0145171236c85c2cf811d3524c0f124ac6",
      "raw_data_url": "file:///home/user/workspace/factuality_bench/amplify_v2_n200_bic.json",
      "record_id": "trust_ebc40d28cff14c7c92a147c765ff59cd",
      "result_status": "preliminary",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "fade22cd6d9ecb6bbfb75b574e0fe9cbbe02e154f7ca026568a557851e84f21f68d8a1d764cb47263aa7f98a870bbb0ea01d09f1b41d07a151fa1ee8e91cca0f",
      "target_compression_ratio": null,
      "task": "factuality_scoring",
      "ties": 208,
      "timestamp_utc": "2026-06-03T15:49:38.062588+00:00",
      "wins": 12
    },
    {
      "adversary": "LLMLingua-2 (microsoft/llmlingua-2-bert-base-multilingual-cased-meetingbank, rate=0.5)",
      "adversary_citation": "Pan et al., LLMLingua-2: Data Distillation for Efficient and Faithful Task-Agnostic Prompt Compression, ACL 2024",
      "adversary_mean": 0.38916,
      "bootstrap_ci_95": [
        0.585,
        0.636
      ],
      "cohen_d": 1.7717997444587166,
      "dataset": "enterprise_casb_real (n=500) + cloudflare_workers_ai (n=500), seed=42",
      "losses": 0,
      "mean_delta": 0.61084,
      "methodology_url": "https://thehiveryiq.com/smsh/",
      "metric": "Invariant recall (1.0 = all policy codes, DIDs, identifiers preserved; lossless)",
      "n": 1000,
      "notes": "SMSH v5 registry+structural is LOSSLESS by construction: 1000/1000 prompts retain all invariants (policy codes, DIDs, identifiers, token IDs). LLMLingua-2 at rate=0.5 retains invariants on 38.9% overall \u2014 drops policy codes on 99.97% of enterprise CASB prompts (inv 0.00032), 77.8% on cloudflare. SMSH compress p50 0.81ms vs LL2 1638ms. SMSH compression 1.53x vs LL2 1.85x (LL2 wins raw ratio but drops required structure). Adversary: Pan et al. ACL 2024.",
      "p_value_two_sided_paired_t": 0.0,
      "paired_t_statistic": 39.85,
      "primitive": "smsh",
      "primitive_mean": 1.0,
      "primitive_version": "v5-registry",
      "pubkey_hex": "fd0340a905cd712c1cdd08764fcbfcc3822f406afc8011fb320f0ba4ad834a25",
      "raw_data_url": "https://thehiveryiq.com/assets/trust-bootstrap.json",
      "record_id": "trust_5bd773a9d5c0447b887e7cdd968fec4e",
      "result_status": "publishable",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "805ce3c7df61731839015a833b769ed378a6ad9e111838a67b6c04535c066d3e3a636448adf87b15c3526431c4b19e654d3675fd6df0fccf77ebce49aa26a709",
      "target_compression_ratio": null,
      "task": "invariant_preservation_binary",
      "ties": 389,
      "timestamp_utc": "2026-06-03T15:52:42.183720+00:00",
      "wins": 611
    },
    {
      "adversary": "protectai/deberta-v3-base-prompt-injection-v2",
      "adversary_citation": "ProtectAI, DeBERTa-v3 Prompt Injection v2 (HuggingFace, 2024)",
      "adversary_mean": 0.988,
      "bootstrap_ci_95": [
        0.003,
        0.025
      ],
      "cohen_d": 0.1558573000398394,
      "dataset": "lmsys/toxic-chat + jackhhao/jailbreak-classification + deepset/prompt-injections, balanced 500/500",
      "losses": 0,
      "mean_delta": 0.012,
      "methodology_url": "https://thehiveryiq.com/xcalibur/membrane/",
      "metric": "Specificity on benign set (1.0 = zero false positives; precision-first)",
      "n": 500,
      "notes": "MEMBRANE v2 at zero-false-positive threshold: 500/500 benign prompts pass cleanly (Precision=1.000, FPR=0.000). DeBERTa-PI-v2 flags 6/500 benign as adversarial (Precision=0.967, FPR=0.012). On full balanced corpus n=1000, DeBERTa wins recall-driven Macro-F1 (0.629 vs 0.445); MEMBRANE wins precision-first ops (zero benign-traffic interference). MEMBRANE recall=0.110 \u2014 pair with recall-tuned layer for defense-in-depth.",
      "p_value_two_sided_paired_t": 0.000525820173219893,
      "paired_t_statistic": 3.49,
      "primitive": "membrane",
      "primitive_mean": 1.0,
      "primitive_version": "v2-precision-mode",
      "pubkey_hex": "fd0340a905cd712c1cdd08764fcbfcc3822f406afc8011fb320f0ba4ad834a25",
      "raw_data_url": "https://thehiveryiq.com/assets/trust-bootstrap.json",
      "record_id": "trust_720293dc93e34baa92db05f9b3c1233d",
      "result_status": "preliminary",
      "schema_version": "hive.trust.benchmark.v1",
      "signature": "0323b6d20d2f28ab8d46f02f46b244c0a35ed874c738d3d081c838791c04f9461177a31586e50c5ef143216793ce258a8b8e9a8f83ca3668186ffb1767b16309",
      "target_compression_ratio": null,
      "task": "specificity_at_zero_fpr_operating_point",
      "ties": 494,
      "timestamp_utc": "2026-06-03T15:52:42.184747+00:00",
      "wins": 6
    },
    {
      "schema_version": "hive.trust.benchmark.v1",
      "primitive": "deltarecon5",
      "primitive_version": "v2",
      "primitive_mean": 1.0,
      "adversary": "Unsigned divergence float (no receipt, no tamper evidence)",
      "adversary_citation": "Any system returning semantic divergence as an unsigned float \u2014 cannot detect corpus swap. Represents Together.ai, Groq, Anyscale delta outputs.",
      "adversary_mean": 0.0,
      "metric": "tamper_detection_rate (1.0=catches every corpus swap, 0.0=catches none)",
      "dataset": "Hive VoiceRecon1 corpus v1 \u2014 16 public-record passages, 500 tamper trials",
      "dataset_provenance": "500 trials: 250 tampered (silent corpus swap +0.15 divergence mutation), 250 clean. Passages from public earnings calls and legal depositions.",
      "task": "counterfactual_tamper_detection",
      "n": 500,
      "wins": 250,
      "losses": 0,
      "ties": 250,
      "cohen_d": 1.4128,
      "mean_delta": 1.0,
      "p_value_two_sided_paired_t": 0.0,
      "paired_t_statistic": null,
      "bootstrap_ci_95": [
        1.0,
        1.0
      ],
      "tamper_detection_rate": 1.0,
      "result_status": "preliminary",
      "record_id": "trust_dr5v2_fec806f8612a940a5eadb10336c2e888",
      "pubkey_hex": "9e883b3dfcf7ecc9125b09b4b594f74b611d8913c38e5b72d9a4faacfbafc9d6",
      "methodology_url": "https://thehiveryiq.com/xcalibur/delta/",
      "timestamp_utc": "2026-06-04T12:09:56.959904Z",
      "notes": "DR5 v2: correct adversary = unsigned system that cannot detect corpus swaps. Hive DR5 catches 250/250 tampered trials via Ed25519 sig mismatch. Unsigned baseline catches 0/250. Cohen d=1.413.",
      "target_compression_ratio": null
    },
    {
      "schema_version": "hive.trust.benchmark.v1",
      "primitive": "voicerecon1",
      "primitive_version": "v2",
      "primitive_mean": 0.80737,
      "adversary": "Naive head-truncation (same word count, no semantic selection)",
      "adversary_citation": "Head truncation: first N words of original. Represents any system clipping context without semantic scoring.",
      "adversary_mean": 0.657949,
      "metric": "invariant_recall (cosine sim vs original, higher=better, 0-1)",
      "dataset": "Hive VoiceRecon1 corpus v1 \u2014 8 earnings calls + 8 legal depositions (public record)",
      "dataset_provenance": "Verbatim passages from public SEC filings, investor relations transcripts, and public court records: NVIDIA Q4 FY2024, MSFT Q3 FY2024, Apple Q2 FY2024, Alphabet Q1 2024, Meta Q1 2024, Amazon Q1 2024, Salesforce Q4 FY2024, JPMorgan Q1 2024; FTC v. Meta, SEC v. Ripple, AI patent litigation, healthcare breach, financial fraud, CFTC v. FTX, SEC v. Terraform depositions.",
      "task": "voice_smsh_compression_invariant_recall",
      "n": 48,
      "wins": 41,
      "losses": 7,
      "ties": 0,
      "cohen_d": 0.907071,
      "mean_delta": 0.149421,
      "compression_by_target": {
        "3x": {
          "mean_smsh_recall": 0.858043,
          "mean_trunc_recall": 0.75689,
          "mean_actual_ratio": 2.26,
          "cohen_d": 0.717741
        },
        "5x": {
          "mean_smsh_recall": 0.805592,
          "mean_trunc_recall": 0.675035,
          "mean_actual_ratio": 2.92,
          "cohen_d": 0.834254
        },
        "9x": {
          "mean_smsh_recall": 0.758474,
          "mean_trunc_recall": 0.541922,
          "mean_actual_ratio": 4.283,
          "cohen_d": 1.356921
        }
      },
      "dr5_tamper_detection": {
        "n_trials": 500,
        "n_tampered": 250,
        "hive_dr5_tdr": 1.0,
        "unsigned_base_tdr": 0.0,
        "cohen_d": 1.412799,
        "description": "Hive DR5 detects 100% of corpus swaps via Ed25519 sig mismatch. Unsigned baseline detects 0%."
      },
      "stt_accuracy": {
        "mean_wer": 0.145413,
        "accuracy": 0.854587,
        "cohen_d": 4.278009,
        "n_passages": 16
      },
      "result_status": "preliminary",
      "record_id": "trust_vr1_fec806f8612a940a5eadb10336c2e888",
      "pubkey_hex": "9e883b3dfcf7ecc9125b09b4b594f74b611d8913c38e5b72d9a4faacfbafc9d6",
      "receipts_count": 48,
      "methodology_url": "https://thehiveryiq.com/xcalibur/delta/",
      "timestamp_utc": "2026-06-04T11:58:29.107115+00:00",
      "notes": "SMSH sliding-window semantic chunker v1. Corpus: 16 real public-record passages (16 total). Adversary: head truncation (not LLM-based). DR5 tamper detection: Hive=1.000 vs unsigned=0.000 on 500 trials. STT WER=0.1454 on paraphrase proxy (full audio run needs TTS key). All 48 compression receipts Ed25519 signed.",
      "target_compression_ratio": null
    }
  ]
}