{"id":"W4412584687","doi":"10.1016/j.media.2025.103716","title":"PitVis-2023 challenge: Workflow recognition in videos of endoscopic pituitary surgery","year":2025,"lang":"en","type":"article","venue":"Medical Image Analysis","topic":"Surgical Simulation and Training","field":"Medicine","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Department for Science, Innovation and Technology; Horizon 2020 Framework Programme; Cancer Research UK; Wellcome / EPSRC Centre for Interventional and Surgical Sciences; Wellcome Trust; Royal Academy of Engineering; Engineering and Physical Sciences Research Council; National Institute for Health and Care Research","keywords":"Workflow; Computer science; Artificial intelligence; Database","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004042927,0.004922586,0.001940607,0.001648723,0.001503752,0.00298954,0.003058856,0.004315448,0.01842571],"category_scores_gemma":[0.01490169,0.0006542372,0.00256026,0.001742381,0.0008185375,0.002879012,0.003203637,0.003559376,0.01442396],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002817364,"about_ca_system_score_gemma":0.004017162,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03068032,"about_ca_topic_score_gemma":0.07006395,"domain_scores_codex":[0.9955226,0.001194147,0.0002876628,0.001132318,0.001271605,0.0005916861],"domain_scores_gemma":[0.9937728,0.001921753,0.0002474697,0.001099652,0.001923262,0.001035006],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001781671,0.0007621541,0.001908391,0.002501464,0.0002997986,0.0007019929,0.0002291743,0.01078073,0.008521093,0.001206443,0.801936,0.169371],"study_design_scores_gemma":[0.001281265,0.002930399,0.02858119,0.001726344,0.0003014499,0.003806356,0.00182058,0.2998036,0.03543251,0.01353506,0.6102394,0.0005418176],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.1823561,0.026451,0.1142855,0.01419823,0.03023793,0.006131311,0.4754994,0.08825156,0.06258889],"genre_scores_gemma":[0.1284492,0.003059089,0.0940937,0.002026281,0.001756707,0.001294011,0.7218756,0.003037113,0.04440834],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03068032,"threshold_uncertainty_score":0.06164014,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03061313576358627,"score_gpt":0.3268501300808107,"score_spread":0.2962369943172244,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}