{"id":"W4411515984","doi":"10.1002/sim.70176","title":"Precision of Treatment Hierarchy: A Metric for Quantifying Certainty in Treatment Hierarchies From Network Meta‐Analysis","year":2025,"lang":"en","type":"article","venue":"Statistics in Medicine","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Actua; University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; Deutsche Forschungsgemeinschaft; Hellenic Foundation for Research and Innovation; European Commission","keywords":"Metric (unit); Ranking (information retrieval); Hierarchy; Pairwise comparison; Statistics; Mathematics; Frequentist inference; Rank (graph theory); Certainty; Variance (accounting); Similarity (geometry); Computer science; Bayesian probability; Econometrics; Artificial intelligence; Bayesian inference","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.03721322,0.0004102873,0.009117238,0.003366645,0.0001229426,0.00007847154,0.0008225415,0.00008653013,0.00236469],"category_scores_gemma":[0.02840004,0.0001812715,0.001692648,0.008698816,0.000143877,0.00005539501,0.0001252009,0.0001002467,0.00001338118],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003007835,"about_ca_system_score_gemma":0.0001388753,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003861987,"about_ca_topic_score_gemma":0.01261251,"domain_scores_codex":[0.9816496,0.005788522,0.008967068,0.001084807,0.002141821,0.0003681823],"domain_scores_gemma":[0.9440317,0.05086654,0.002332345,0.002140083,0.000529454,0.00009988371],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004828739,0.0008966347,0.2921605,0.0002584317,0.1879778,0.00005409306,0.009528711,0.0714571,0.00003913403,0.09625768,0.01839619,0.3224909],"study_design_scores_gemma":[0.003657789,0.001312294,0.06183621,0.0001680392,0.1813556,5.714864e-7,0.001901002,0.2541434,0.00003253012,0.4768298,0.01840185,0.0003608699],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01849564,0.02630333,0.9492127,0.000972563,0.0003786469,0.002415522,0.000746447,0.000005124647,0.001470012],"genre_scores_gemma":[0.8070257,0.001517992,0.1862488,0.0001265706,0.00008564591,0.0004705967,0.0002518017,0.00001402098,0.004258879],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.7885301,"threshold_uncertainty_score":0.9985473,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7096301178491243,"score_gpt":0.5685468593222872,"score_spread":0.1410832585268371,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}