{"id":"W4410785635","doi":"10.3389/frai.2025.1592399","title":"Moving LLM evaluation forward: lessons from human judgment research","year":2025,"lang":"en","type":"article","venue":"Frontiers in Artificial Intelligence","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Quebec - Clinical Research Organization in Cancer","funders":"","keywords":"Parallels; Management science; Engineering ethics; Computer science; Psychology; Epistemology; Sociology; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003345301,0.0001415124,0.0002125028,0.0006357428,0.0003450403,0.0002878422,0.001404468,0.0001171951,0.00003720884],"category_scores_gemma":[0.0004287686,0.0001568495,0.00005989712,0.001121577,0.0001132696,0.0003659818,0.0004895608,0.0004545588,0.00004172218],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005525519,"about_ca_system_score_gemma":0.0002546663,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008838606,"about_ca_topic_score_gemma":0.0005030488,"domain_scores_codex":[0.9969365,0.0004031895,0.0005862277,0.0007576602,0.0008159683,0.0005004373],"domain_scores_gemma":[0.9985533,0.0001714137,0.00007108183,0.0008757993,0.0002595914,0.00006879073],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000009011639,0.0001192713,0.001264887,0.000009086493,0.0000153265,0.000005952491,0.00181053,0.01146107,0.001522073,0.2462028,0.002198277,0.7353817],"study_design_scores_gemma":[0.00002340656,0.00001633367,0.0002336476,0.0000729522,0.000004015406,8.7388e-8,0.0006744341,0.5314249,0.01540643,0.4516006,0.0004503584,0.00009281265],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03040504,0.0005747114,0.9596327,0.002926298,0.002274962,0.0004410901,0.000002055211,0.00007367762,0.003669489],"genre_scores_gemma":[0.8918556,0.00002499535,0.1075386,0.0001337189,0.000106201,0.00009295092,0.000004965597,0.000007304457,0.0002356928],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8614506,"threshold_uncertainty_score":0.6396136,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1868207403157975,"score_gpt":0.4347521440215913,"score_spread":0.2479314037057938,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}