{"id":"W4417274551","doi":"10.2196/77707","title":"Information Extraction of Doctoral Theses Using Two Different Large Language Models vs Health Services Researchers: Development and Usability Study","year":2025,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Workflow; Usability; Information extraction; Information system; Health information; Doctoral dissertation; MEDLINE","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02461585,0.001530794,0.001408143,0.003854351,0.0006068447,0.002738727,0.00111384,0.001152417,0.002513705],"category_scores_gemma":[0.09091224,0.0005662676,0.002145731,0.002324406,0.0005240146,0.003972027,0.00296029,0.001416596,0.001010928],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001492918,"about_ca_system_score_gemma":0.001737369,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002485088,"about_ca_topic_score_gemma":0.002587765,"domain_scores_codex":[0.9842126,0.01100324,0.002026154,0.001546301,0.000978852,0.0002329361],"domain_scores_gemma":[0.848905,0.1392386,0.002481651,0.00321866,0.00531184,0.0008441341],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.01240325,0.007103238,0.09683489,0.02509174,0.002492889,0.001940241,0.05016237,0.0106583,0.02186818,0.002608899,0.0288309,0.7400051],"study_design_scores_gemma":[0.00816046,0.02384048,0.2771254,0.005798028,0.007075492,0.003855838,0.05543336,0.4408496,0.06767703,0.01180138,0.09677702,0.001605997],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9253514,0.001649515,0.04883302,0.0008797566,0.0001732988,0.007209798,0.007267342,0.006093873,0.002542031],"genre_scores_gemma":[0.7624942,0.001176057,0.2053011,0.0006124545,0.0001093814,0.010681,0.01712158,0.0006009922,0.001903276],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9753842,"threshold_uncertainty_score":0.1301826,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3894185055392182,"score_gpt":0.5962567410266861,"score_spread":0.2068382354874679,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}