{"id":"W7077860944","doi":"10.48448/e7e3-kg06","title":"Rethinking Safety Evaluation in Large Language Models: A Research Proposal","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Occupational safety and health; Risk assessment; Health care; Public health; Entertainment; Robustness (evolution)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01350622,0.0002075898,0.0002470445,0.001138733,0.0003305376,0.0002441426,0.002628693,0.0002852484,0.0003080296],"category_scores_gemma":[0.001294902,0.0001879036,0.00003618199,0.002965688,0.0004100103,0.0003608693,0.001404058,0.0008629306,0.0000486808],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003484872,"about_ca_system_score_gemma":0.003631381,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005297427,"about_ca_topic_score_gemma":0.001275713,"domain_scores_codex":[0.9955542,0.000340635,0.0003260139,0.001086596,0.001876794,0.0008157816],"domain_scores_gemma":[0.9980049,0.0001438797,0.0001349853,0.00108455,0.0005414287,0.00009023461],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001229108,0.0003198436,0.0001241506,0.0003142099,0.00002114732,0.0001046887,0.008663405,0.00430586,0.0008979894,0.863058,0.04137417,0.08080431],"study_design_scores_gemma":[0.0003496989,0.00002354241,0.00002368483,0.0004651084,0.000003917382,0.000004664848,0.0003220686,0.7987299,0.0002029935,0.1789902,0.02065048,0.0002337254],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.00002923299,0.0005366408,0.09646413,0.003299243,0.0003401489,0.0006452274,0.000009623411,0.0002230724,0.8984527],"genre_scores_gemma":[0.2381251,0.0001112429,0.1074356,0.000475088,0.0003936731,0.000117901,0.00008517799,0.00004832583,0.6532079],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.794424,"threshold_uncertainty_score":0.7662485,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0685846569655526,"score_gpt":0.3647848096916979,"score_spread":0.2962001527261453,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}