{
 "schema": "gaige-receipt-export/1",
 "exported_by": "gaige 0.0.2",
 "receipt": {
  "id": "20260722-213952-binoculars",
  "generated_utc": "2026-07-23T02:40:53+00:00",
  "gaige_version": "0.0.1"
 },
 "instrument": {
  "host": {
   "os": "Linux",
   "arch": "x86_64",
   "device": "cuda"
  },
  "detector": {
   "detector": "binoculars",
   "paper": "Hans et al., Binoculars (2024), arXiv:2401.12070; released-implementation construction",
   "model_id": "tiiuae/falcon-7b + tiiuae/falcon-7b-instruct",
   "observer_id": "tiiuae/falcon-7b",
   "performer_id": "tiiuae/falcon-7b-instruct",
   "quant_requested": "4bit",
   "quant_verified": {
    "linear4bit_modules": 256,
    "resident_gb": 8.07,
    "per_model": {
     "observer": {
      "linear4bit_modules": 128,
      "resident_gb": 4.04
     },
     "performer": {
      "linear4bit_modules": 128,
      "resident_gb": 4.03
     }
    }
   },
   "max_tokens": 1024,
   "versions": {
    "torch": "2.13.0+cu130",
    "transformers": "4.49.0",
    "cuda": "13.0",
    "python": "3.12.3",
    "bitsandbytes": "0.49.2"
   },
   "device": "cuda",
   "device_requested": "cuda",
   "device_fallback": false,
   "model_auto_selected": false,
   "compute": {
    "name": "NVIDIA RTX 5000 Ada Generation Laptop GPU"
   },
   "score_semantics": "NEGATED Binoculars ratio -(ppl/x_ppl); higher = more AI-like; RAW criterion (uncalibrated by design; the paper's global thresholds are deliberately not used)"
  }
 },
 "corpus": {
  "name": "hc3-mini(n=100,seed=17)",
  "sha256": "7d2819d3e83bd10dc3cef56b1fcd2b19f09ada6ef59d671b25498635f6937d01",
  "counts": {
   "human": 100,
   "ai": 100
  },
  "meta": {
   "source": "HC3 (Hello-SimpleAI) via HF hub",
   "url": "https://huggingface.co/datasets/Hello-SimpleAI/HC3/resolve/main/all.jsonl",
   "raw_sha256": "ee231f82283d754050a4dc58ecf9331afaf5daea7283c6da0b096ed0fb2774cf",
   "filters": {
    "min_words": 50,
    "max_words": 300
   },
   "n_per_class": 100,
   "seed": 17,
   "note": "known-AI side is ChatGPT-era text; detectors may score newer model families differently"
  }
 },
 "metrics": {
  "auroc": 0.9992,
  "auroc_ci": [
   0.9974,
   1.0
  ],
  "n_boot": 1000
 },
 "thresholds": [
  {
   "target_fpr": 0.01,
   "threshold": -0.7828962206840515,
   "achieved_fpr": 0.01,
   "achieved_tpr": 0.97,
   "tpr_ci": [
    0.94,
    1.0
   ]
  },
  {
   "target_fpr": 0.05,
   "threshold": -0.8706418871879578,
   "achieved_fpr": 0.05,
   "achieved_tpr": 1.0,
   "tpr_ci": [
    1.0,
    1.0
   ]
  }
 ],
 "conformal": [
  {
   "alpha": 0.05,
   "threshold": -0.8706418871879577,
   "n_calibration": 100,
   "order_statistic": 96,
   "conditional_fpr_mean": 0.04950495049504951,
   "conditional_fpr_sd": 0.021478263150361998,
   "guarantee": "P(human flagged) <= 0.05, marginal over calibration draws, under exchangeability (split conformal)",
   "tpr": 1.0
  },
  {
   "alpha": 0.01,
   "threshold": -0.7540451884269713,
   "n_calibration": 100,
   "order_statistic": 100,
   "conditional_fpr_mean": 0.009900990099009901,
   "conditional_fpr_sd": 0.009803441019571034,
   "guarantee": "P(human flagged) <= 0.01, marginal over calibration draws, under exchangeability (split conformal)",
   "tpr": 0.95
  },
  {
   "alpha": 0.005,
   "unavailable": "alpha=0.005 needs >= 199 human calibration samples, got 100. A tighter guarantee than your data supports is not a guarantee."
  }
 ],
 "subgroups": {
  "min_subgroup": 20,
  "by_threshold": [
   {
    "target_fpr": 0.01,
    "threshold": -0.7828962206840515,
    "strata": {
     "length_bucket": {
      "0-100w": {
       "n_human": 42,
       "n_ai": 7,
       "fpr": 0.023809523809523808,
       "fpr_ci": [
        0.0,
        0.07142857142857142
       ],
       "tpr": null,
       "tpr_ci": null,
       "rate_withheld": true
      },
      "100-250w": {
       "n_human": 39,
       "n_ai": 81,
       "fpr": 0.0,
       "fpr_ci": [
        0.0,
        0.0
       ],
       "tpr": 0.9629629629629629,
       "tpr_ci": [
        0.9259259259259259,
        1.0
       ],
       "rate_withheld": false
      },
      "250-500w": {
       "n_human": 19,
       "n_ai": 12,
       "fpr": null,
       "fpr_ci": null,
       "tpr": null,
       "tpr_ci": null,
       "rate_withheld": true
      }
     }
    },
    "max_fpr_disparity": {
     "length_bucket": {
      "gap": 0.023809523809523808,
      "worst_group": "0-100w",
      "worst_fpr": 0.023809523809523808,
      "best_group": "100-250w",
      "best_fpr": 0.0,
      "ratio": null
     }
    }
   },
   {
    "target_fpr": 0.05,
    "threshold": -0.8706418871879578,
    "strata": {
     "length_bucket": {
      "0-100w": {
       "n_human": 42,
       "n_ai": 7,
       "fpr": 0.09523809523809523,
       "fpr_ci": [
        0.023809523809523808,
        0.19047619047619047
       ],
       "tpr": null,
       "tpr_ci": null,
       "rate_withheld": true
      },
      "100-250w": {
       "n_human": 39,
       "n_ai": 81,
       "fpr": 0.02564102564102564,
       "fpr_ci": [
        0.0,
        0.07692307692307693
       ],
       "tpr": 1.0,
       "tpr_ci": [
        1.0,
        1.0
       ],
       "rate_withheld": false
      },
      "250-500w": {
       "n_human": 19,
       "n_ai": 12,
       "fpr": null,
       "fpr_ci": null,
       "tpr": null,
       "tpr_ci": null,
       "rate_withheld": true
      }
     }
    },
    "max_fpr_disparity": {
     "length_bucket": {
      "gap": 0.0695970695970696,
      "worst_group": "0-100w",
      "worst_fpr": 0.09523809523809523,
      "best_group": "100-250w",
      "best_fpr": 0.02564102564102564,
      "ratio": 3.7142857142857144
     }
    }
   }
  ]
 },
 "base_rate": {
  "volume": 75000,
  "volume_note": "default is Vanderbilt's published 75,000 submissions/year",
  "at": [
   {
    "target_fpr": 0.01,
    "expected_false_positives": 750.0,
    "ppv_at_prevalence": [
     {
      "prevalence": 0.01,
      "ppv": 0.4948979591836735
     },
     {
      "prevalence": 0.1,
      "ppv": 0.9150943396226414
     },
     {
      "prevalence": 0.5,
      "ppv": 0.9897959183673469
     }
    ]
   },
   {
    "target_fpr": 0.05,
    "expected_false_positives": 3750.0,
    "ppv_at_prevalence": [
     {
      "prevalence": 0.01,
      "ppv": 0.1680672268907563
     },
     {
      "prevalence": 0.1,
      "ppv": 0.689655172413793
     },
     {
      "prevalence": 0.5,
      "ppv": 0.9523809523809523
     }
    ]
   }
  ]
 },
 "reproduce": {
  "run": "gaige run --corpus hc3-mini --n 100 --seed 17 --detector binoculars --observer tiiuae/falcon-7b --performer tiiuae/falcon-7b-instruct --quant 4bit --device cuda --max-tokens 1024"
 }
}