{"id":"W4412584687","doi":"10.1016/j.media.2025.103716","title":"PitVis-2023 challenge: Workflow recognition in videos of endoscopic pituitary surgery","year":2025,"lang":"en","type":"article","venue":"Medical Image Analysis","topic":"Surgical Simulation and Training","field":"Medicine","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Department for Science, Innovation and Technology; Horizon 2020 Framework Programme; Cancer Research UK; Wellcome / EPSRC Centre for Interventional and Surgical Sciences; Wellcome Trust; Royal Academy of Engineering; Engineering and Physical Sciences Research Council; National Institute for Health and Care Research","keywords":"Workflow; Computer science; Artificial intelligence; Database","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0009029991,0.0001194795,0.0007306468,0.001028683,0.00002726854,0.000008166258,0.00007256446,0.0001535393,0.00682757],"category_scores_gemma":[0.002357758,0.0001025765,0.0003900865,0.002772038,0.0001159863,0.00007080133,0.00003636066,0.0002862745,0.00004493837],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004109781,"about_ca_system_score_gemma":0.0001421232,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001781808,"about_ca_topic_score_gemma":0.0001968437,"domain_scores_codex":[0.9981609,0.0001427306,0.0006446802,0.0002780915,0.0005467911,0.0002268335],"domain_scores_gemma":[0.9981956,0.001202134,0.00008088249,0.0002270996,0.0001142281,0.0001799853],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0002108007,0.001056181,0.2657098,0.0002765122,0.001875425,0.0008025314,0.0001862864,0.00003164102,0.00007564154,0.00004644766,0.0006262845,0.7291024],"study_design_scores_gemma":[0.01092921,0.0001946082,0.9037489,0.003115189,0.006296695,0.00000647827,0.0008739514,0.0637821,0.001488688,0.003075579,0.005948785,0.000539812],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9446583,0.001667248,0.009800768,0.009190002,0.0002305532,0.0002825114,0.00001052058,0.000104474,0.03405563],"genre_scores_gemma":[0.9976521,0.0004156271,0.0004176515,0.0009472331,0.00006837327,0.00001764496,0.0001584118,0.000007238903,0.0003157785],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7285627,"threshold_uncertainty_score":0.9940803,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03061313576358627,"score_gpt":0.3268501300808107,"score_spread":0.2962369943172244,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}