{"id":"W4405510733","doi":"10.2196/69830","title":"Peer Review of “Towards Evaluating the Diagnostic Ability of LLMs (Preprint)”","year":2024,"lang":"en","type":"article","venue":"JMIRx Med","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"","keywords":"Preprint; Peer review; Political science; Computer science; World Wide Web; Law","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.003896992,0.0000660834,0.0002265077,0.0000329839,0.00002741589,0.000005009914,0.00008708306,0.00004588093,0.0008047849],"category_scores_gemma":[0.02711599,0.0000405507,0.0001083754,0.0002746223,0.00009372858,0.00003273122,0.00002676899,0.0001799154,0.00006339241],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005815056,"about_ca_system_score_gemma":0.0004747023,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005229457,"about_ca_topic_score_gemma":0.00001004561,"domain_scores_codex":[0.9984117,0.0001324318,0.0005636441,0.0001662193,0.0006118197,0.0001142548],"domain_scores_gemma":[0.9966605,0.001588602,0.00009642904,0.0004400762,0.001157928,0.00005650433],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00004931926,0.0001838457,0.007243861,0.0317504,0.00005830506,0.000003218408,0.008015889,0.000008828733,0.006383185,0.0007210632,0.02933525,0.9162468],"study_design_scores_gemma":[0.0002410951,0.003792784,0.2932095,0.1477733,0.001824955,0.000116344,0.01121356,0.01238704,0.3218058,0.03343201,0.1735053,0.0006983599],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8163479,0.02433726,0.0001651397,0.153126,0.0009781522,0.001589503,0.000006671289,0.00004359765,0.003405791],"genre_scores_gemma":[0.9950125,0.002358377,0.0001917726,0.0005335457,0.0001914402,0.000122155,0.0000117674,0.000008739412,0.001569664],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9155485,"threshold_uncertainty_score":0.981079,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3319253764091181,"score_gpt":0.5564798883454766,"score_spread":0.2245545119363584,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}