{"id":"W6902327585","doi":"10.6084/m9.figshare.25801054.v1","title":"Additional file 1 of An evaluation of computational methods for aggregate data meta-analyses of diagnostic test accuracy studies","year":2024,"lang":"en","type":"article","venue":"Figshare","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Aggregate (composite); Diagnostic test; Diagnostic accuracy; Test data; Test (biology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":["insufficient_payload"],"domain":null,"study_design":"not_applicable","genre":"other","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"},{"model":"gpt","categories":["insufficient_payload"],"domain":null,"study_design":"not_applicable","genre":"dataset","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"}],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.01806291,0.001715631,0.001536916,0.002902941,0.0007720995,0.001882224,0.00301418,0.001997408,0.8442824],"category_scores_gemma":[0.2051503,0.001448691,0.003360952,0.004517294,0.0003852755,0.002078776,0.001428557,0.001899197,0.08716194],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002064778,"about_ca_system_score_gemma":0.003386286,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005662075,"about_ca_topic_score_gemma":0.009330654,"domain_scores_codex":[0.9933053,0.003993348,0.000898215,0.0005948859,0.0009337233,0.000274589],"domain_scores_gemma":[0.6567594,0.3226241,0.005376302,0.005424719,0.008917801,0.0008977002],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00257194,0.0002923101,0.002994099,0.0149259,0.0009956772,0.0001123135,0.0001449389,0.01290499,0.0001204238,0.005303484,0.9354144,0.02421956],"study_design_scores_gemma":[0.07908735,0.002284025,0.02627428,0.01804567,0.00547234,0.001232868,0.0004631499,0.07935924,0.002614617,0.08180042,0.7028847,0.0004814394],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"methods","genre_scores_codex":[0.0006498061,0.00009177245,0.005101544,0.0003171001,0.00006180932,0.001077377,0.9887018,0.00107538,0.002923402],"genre_scores_gemma":[0.06461708,0.0009162026,0.1064912,0.002743462,0.0004436688,0.04533489,0.7481788,0.008257813,0.02301692],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9819371,"threshold_uncertainty_score":0.2221122,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9716713284645822,"score_gpt":0.7108307869855779,"score_spread":0.2608405414790043,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}