{"id":"W4410785635","doi":"10.3389/frai.2025.1592399","title":"Moving LLM evaluation forward: lessons from human judgment research","year":2025,"lang":"en","type":"article","venue":"Frontiers in Artificial Intelligence","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Quebec - Clinical Research Organization in Cancer","funders":"","keywords":"Parallels; Management science; Engineering ethics; Computer science; Psychology; Epistemology; Sociology; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1428725,0.001691393,0.002068677,0.006924564,0.002395486,0.01554692,0.004619963,0.00416481,0.006021155],"category_scores_gemma":[0.4514498,0.0007202115,0.001078405,0.003894262,0.01131246,0.03141399,0.00900379,0.007183075,0.001634774],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006075396,"about_ca_system_score_gemma":0.006867742,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009757768,"about_ca_topic_score_gemma":0.008562386,"domain_scores_codex":[0.861655,0.1052957,0.004736118,0.006291251,0.02072189,0.001300061],"domain_scores_gemma":[0.5628392,0.3530945,0.01190631,0.02804313,0.03956738,0.00454955],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0005944991,0.0005121577,0.01849711,0.002341173,0.0004861212,0.0002248242,0.0227003,0.01709274,0.001478905,0.3057813,0.02161928,0.6086715],"study_design_scores_gemma":[0.00009737573,0.0002629727,0.004502897,0.001666318,0.000079777,0.0001321326,0.006733189,0.04482511,0.00197103,0.908754,0.03075978,0.0002154226],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.07365201,0.02486745,0.7448116,0.0996585,0.001541399,0.0005498113,0.0004738648,0.001524229,0.05292115],"genre_scores_gemma":[0.7317971,0.004835928,0.2537951,0.005314744,0.0009574151,0.0003591093,0.0003562477,0.0005794124,0.00200485],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8571275,"threshold_uncertainty_score":0.7555912,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1868207403157975,"score_gpt":0.4347521440215913,"score_spread":0.2479314037057938,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}