{"id":"W4410995591","doi":"10.1136/bmj.r1105","title":"Efficiency must not compromise trustworthiness in rating certainty and formulating recommendations in AI era","year":2025,"lang":"en","type":"editorial","venue":"BMJ","topic":"Cardiac, Anesthesia and Surgical Outcomes","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McMaster University; Impact","funders":"","keywords":"Compromise; Trustworthiness; Certainty; Computer science; Rating system; Data science; Operations research; Internet privacy; Political science; Mathematics; Economics; Environmental economics; Law","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1136203,0.003010478,0.006695365,0.00719327,0.004098409,0.01581283,0.005469861,0.0242038,0.009803246],"category_scores_gemma":[0.5393347,0.002677388,0.004120463,0.00426426,0.006766988,0.009727364,0.003321369,0.03553953,0.009794123],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004958051,"about_ca_system_score_gemma":0.01327145,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00477914,"about_ca_topic_score_gemma":0.01436901,"domain_scores_codex":[0.898979,0.03800553,0.02297842,0.003270748,0.03528897,0.001477216],"domain_scores_gemma":[0.4883357,0.3540652,0.02390874,0.01255425,0.11235,0.00878616],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00015023,0.00001848826,0.0001228817,0.003400252,0.0004805193,0.00009073947,0.0001549294,0.00009575264,0.0000762882,0.003327738,0.9609993,0.03108291],"study_design_scores_gemma":[0.0009337693,0.0001017673,0.001016946,0.01066846,0.001373892,0.0002450855,0.0003021845,0.001538184,0.00041239,0.03152083,0.951678,0.0002084355],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"editorial","genre_gemma":"editorial","genre_scores_codex":[0.0001099586,0.01267829,0.003048399,0.1704947,0.8094338,0.0002997543,0.0002720443,0.0001799733,0.003482943],"genre_scores_gemma":[0.004030285,0.01375284,0.01012183,0.1074199,0.8541498,0.000683191,0.0001571177,0.0002399088,0.009445077],"genre_candidate":"editorial","genre_consensus":"editorial","teacher_disagreement_score":0.8863797,"threshold_uncertainty_score":0.6008886,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0139869888662449,"score_gpt":0.3430071219852658,"score_spread":0.329020133119021,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}