{"id":"W4399327635","doi":"10.1016/j.radonc.2024.110345","title":"A joint ESTRO and AAPM guideline for development, clinical validation and reporting of artificial intelligence models in radiation therapy","year":2024,"lang":"en","type":"article","venue":"Radiotherapy and Oncology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":81,"is_retracted":false,"has_abstract":true,"ca_institutions":"Princess Margaret Cancer Centre; University of Toronto; University Health Network","funders":"European SocieTy for Radiotherapy and Oncology","keywords":"Guideline; Quality assurance; Medical physics; Computer science; Pace; Delphi; Process (computing); Delphi method; Quality (philosophy); Management science; Medicine; Process management; Artificial intelligence; Engineering; Pathology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1162926,0.002732764,0.004180098,0.01280395,0.002561939,0.007047109,0.01014664,0.01649432,0.004433892],"category_scores_gemma":[0.206453,0.002540225,0.009894312,0.006489569,0.004153687,0.004947911,0.006678262,0.01110404,0.005187562],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.008754321,"about_ca_system_score_gemma":0.05191445,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0191953,"about_ca_topic_score_gemma":0.02039284,"domain_scores_codex":[0.8561995,0.07166085,0.04269368,0.00236789,0.02432722,0.002750833],"domain_scores_gemma":[0.7641953,0.1131546,0.01589192,0.009062909,0.09345143,0.004243788],"domain_codex":null,"domain_gemma":"reporting","domain_candidate":"reporting","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000205588,0.0005476997,0.001668949,0.02983195,0.0004505818,0.001182912,0.004047478,0.004739191,0.002459345,0.02698596,0.5644986,0.3633819],"study_design_scores_gemma":[0.0003728335,0.0004188281,0.005025103,0.1577063,0.0009979687,0.002023472,0.001786479,0.003981011,0.001865088,0.02507402,0.8004087,0.0003402309],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"methods","genre_scores_codex":[0.007684373,0.1215587,0.3234996,0.3491465,0.02412091,0.06615578,0.01539659,0.003489713,0.08894788],"genre_scores_gemma":[0.02106721,0.06518754,0.7834079,0.04739152,0.002177082,0.04691271,0.0154383,0.0006724393,0.0177453],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8837075,"threshold_uncertainty_score":0.6150211,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4648756613231087,"score_gpt":0.5478378291840205,"score_spread":0.08296216786091176,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}