{"id":"W4412888345","doi":"10.18653/v1/2025.findings-acl.576","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","year":2025,"lang":"en","type":"article","venue":"","topic":"Data Mining Algorithms and Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Generalization; Scalability; Computer science; Artificial intelligence; Machine learning; Mathematics; Database","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004489272,0.00008683202,0.00008536842,0.0001288253,0.0001378941,0.0002471574,0.0001863643,0.00005392242,0.000009808692],"category_scores_gemma":[0.0000533022,0.00008525972,0.00001470027,0.0004835869,0.00001555803,0.0005502977,0.00007659962,0.00004357859,0.000002693418],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006039499,"about_ca_system_score_gemma":0.0001228326,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001824233,"about_ca_topic_score_gemma":0.00007113311,"domain_scores_codex":[0.9990886,0.00003036009,0.0002351685,0.0003558009,0.000145662,0.0001443337],"domain_scores_gemma":[0.9994766,0.00003945028,0.00005653591,0.0002493749,0.0001434066,0.00003462162],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000002702693,0.0001355395,0.003290954,0.00002817882,0.000006975074,4.013823e-7,0.0001057487,0.1373902,0.0009892976,0.02943912,0.002616794,0.8259941],"study_design_scores_gemma":[0.000431525,0.00002305857,0.01401907,0.00002776245,0.00001885951,0.000002149487,0.0000244598,0.9836437,0.0005375727,0.0009716168,0.0002231493,0.0000770375],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.05211895,0.00003351033,0.9460859,0.0002016776,0.0001423792,0.0008920236,0.00001873578,0.0000777419,0.0004291181],"genre_scores_gemma":[0.0687788,0.0000134836,0.9279728,0.0002062494,0.00006162341,0.001927929,0.0002656885,0.000008629525,0.0007648126],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8462536,"threshold_uncertainty_score":0.3476791,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04778180600852341,"score_gpt":0.3289911020724153,"score_spread":0.2812092960638919,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}