{"id":"W4409603033","doi":"10.1101/2025.04.18.25326076","title":"AutoReporter: Development of an artificial intelligence tool for automated assessment of research reporting guideline adherence","year":2025,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia; McMaster University; University of Toronto","funders":"","keywords":"Guideline; Computer science; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03170824,0.00247097,0.001674268,0.006332342,0.0007578802,0.00421282,0.003185751,0.001710593,0.01939874],"category_scores_gemma":[0.1271044,0.001432826,0.002206214,0.002784433,0.0005251935,0.003859831,0.004494215,0.002334092,0.01217801],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002237074,"about_ca_system_score_gemma":0.007156549,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004161878,"about_ca_topic_score_gemma":0.008682119,"domain_scores_codex":[0.9786339,0.009982763,0.004326161,0.003332098,0.003397488,0.0003275729],"domain_scores_gemma":[0.9071506,0.0620863,0.009514914,0.009534247,0.01048037,0.001233629],"domain_codex":null,"domain_gemma":"reporting","domain_candidate":"reporting","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001690296,0.0004518706,0.01087476,0.009045048,0.0007624477,0.0004336117,0.001781907,0.009538149,0.01302746,0.007676693,0.4124842,0.5322335],"study_design_scores_gemma":[0.002042711,0.0009745709,0.01499482,0.003883026,0.0008908217,0.0009344795,0.001278086,0.3867401,0.07186139,0.03719164,0.4785474,0.0006608706],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"software","genre_gemma":"methods","genre_scores_codex":[0.0153739,0.00168633,0.3460979,0.003304256,0.0005848082,0.003699912,0.07739601,0.5461719,0.005684995],"genre_scores_gemma":[0.05983548,0.0007014542,0.8367406,0.001869229,0.0001858553,0.004397304,0.08355428,0.008602614,0.004113152],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9682918,"threshold_uncertainty_score":0.1676912,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1907627597976437,"score_gpt":0.5019322412940268,"score_spread":0.3111694814963831,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}