{"id":"W4405510733","doi":"10.2196/69830","title":"Peer Review of “Towards Evaluating the Diagnostic Ability of LLMs (Preprint)”","year":2024,"lang":"en","type":"article","venue":"JMIRx Med","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"","keywords":"Preprint; Peer review; Political science; Computer science; World Wide Web; Law","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02867677,0.0009350183,0.001573545,0.00575314,0.00569305,0.01011982,0.002709555,0.004718004,0.1873546],"category_scores_gemma":[0.2312177,0.0006343875,0.001569245,0.002259517,0.002965944,0.004454461,0.004844543,0.003487187,0.1454062],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002219746,"about_ca_system_score_gemma":0.01431564,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003514375,"about_ca_topic_score_gemma":0.00501619,"domain_scores_codex":[0.9696391,0.008308209,0.002528164,0.002314758,0.01584163,0.001368128],"domain_scores_gemma":[0.5347226,0.03457787,0.0066032,0.02892581,0.3818044,0.01336602],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002551766,0.0000709793,0.001356943,0.0006072997,0.00005372207,0.0003533124,0.0002748397,0.0001062182,0.002748049,0.001592397,0.9306612,0.06191986],"study_design_scores_gemma":[0.00009694199,0.0001402695,0.005049862,0.0003687816,0.00006988458,0.0004908166,0.0007343526,0.001996623,0.005274182,0.004050058,0.981653,0.00007520939],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"editorial","genre_gemma":"empirical","genre_scores_codex":[0.01997452,0.004181114,0.0310408,0.2211661,0.5887472,0.004287775,0.009663895,0.007935328,0.1130033],"genre_scores_gemma":[0.1235756,0.005337282,0.0378974,0.02641521,0.1406934,0.001939613,0.0187606,0.007292694,0.6380882],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9713233,"threshold_uncertainty_score":0.6267636,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3319253764091181,"score_gpt":0.5564798883454766,"score_spread":0.2245545119363584,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}