{"id":"W4399554391","doi":"10.48550/arxiv.2406.06443","title":"LLM Dataset Inference: Did you train on my dataset?","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Defense Advanced Research Projects Agency; Government of Canada; Canadian Institute for Advanced Research; Alfred P. Sloan Foundation","keywords":"Inference; Computer science; Artificial intelligence; Machine learning; Data mining","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","research_integrity","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0005111638,0.0005512288,0.0004668319,0.0005448807,0.0002442221,0.0003812042,0.004500365,0.0004503303,0.0001474139],"category_scores_gemma":[0.0001658919,0.0006027859,0.0001343088,0.0008563843,0.0001187437,0.000309906,0.007597561,0.002834368,0.001360458],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003119798,"about_ca_system_score_gemma":0.0004748855,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00155202,"about_ca_topic_score_gemma":0.0002695563,"domain_scores_codex":[0.9959509,0.000430751,0.0003402888,0.0024096,0.000255026,0.0006134066],"domain_scores_gemma":[0.995194,0.0003364092,0.0002646548,0.003756792,0.00008116971,0.0003669648],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007420978,0.0004331711,0.001453156,0.001242549,0.0002220748,0.003908461,0.001213126,0.3246256,0.000008514641,0.5278373,0.1307885,0.008193285],"study_design_scores_gemma":[0.0005802893,0.0003810983,0.001048685,0.000564324,0.0001072788,0.00001699717,0.0001016474,0.8087094,0.0000145552,0.06502407,0.12223,0.001221618],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4721,0.0008903402,0.3382424,0.01267206,0.01580846,0.004368089,0.1331056,0.005338446,0.01747454],"genre_scores_gemma":[0.9840268,0.0001266731,0.0008712364,0.0008705159,0.0002179524,0.00000321197,0.01293337,0.00004183048,0.0009083667],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5119268,"threshold_uncertainty_score":0.9996424,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1126358343808321,"score_gpt":0.2624317010573041,"score_spread":0.149795866676472,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}