{"id":"W4221157608","doi":"10.1088/1361-6560/ac8044","title":"OpenKBP-Opt: an international and reproducible evaluation of 76 knowledge-based planning pipelines","year":2022,"lang":"en","type":"article","venue":"Physics in Medicine and Biology","topic":"Advanced Radiotherapy Techniques","field":"Physics and Astronomy","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Princess Margaret Cancer Centre; Vector Institute; University of Toronto","funders":"National Cancer Institute; Natural Sciences and Engineering Research Council of Canada","keywords":"Wilcoxon signed-rank test; Quality assurance; Voxel; Computer science; Test plan; Radiation treatment planning; Reference dose; Pipeline (software); Range (aeronautics); Mathematics; Statistics; Medicine; Artificial intelligence; Radiation therapy; Operations management; Engineering; Surgery","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01961658,0.001715963,0.001068944,0.002518578,0.0009278366,0.002275503,0.003382619,0.001727399,0.00325623],"category_scores_gemma":[0.04263042,0.0009528847,0.001961506,0.002106739,0.001447081,0.002566064,0.004807421,0.001583661,0.0008139469],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004151039,"about_ca_system_score_gemma":0.004468314,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01508619,"about_ca_topic_score_gemma":0.01594483,"domain_scores_codex":[0.9841387,0.008366665,0.001136896,0.001978176,0.003842862,0.0005366526],"domain_scores_gemma":[0.9729875,0.01396975,0.001380309,0.006884689,0.0040423,0.0007353959],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003122162,0.002704575,0.02291512,0.001806738,0.001095936,0.0003177134,0.0009608539,0.5713475,0.01012265,0.004298213,0.01703092,0.3642776],"study_design_scores_gemma":[0.001112177,0.002104287,0.022448,0.0002272183,0.0003067549,0.000267674,0.0005645347,0.9259747,0.02610557,0.004195786,0.01653354,0.0001597248],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6076539,0.001786565,0.318911,0.0008597902,0.0003258831,0.004751884,0.01961473,0.02770133,0.01839487],"genre_scores_gemma":[0.6889399,0.0002317068,0.2845379,0.00018984,0.00003173062,0.002009533,0.02048757,0.002090714,0.001481112],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9803834,"threshold_uncertainty_score":0.1037436,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2866415437114151,"score_gpt":0.4964844889893223,"score_spread":0.2098429452779072,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}