{"id":"W4214926294","doi":"10.1055/s-0042-1743643","title":"Automatic Assessment of Surgical Performance Using Intraoperative Video and Deep Learning: A Comparison with Expert Surgeon Video Review","year":2022,"lang":"en","type":"article","venue":"Journal of Neurological Surgery Part B Skull Base","topic":"Radiology practices and education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Deep learning; Artificial intelligence; Artificial neural network; Medical physics; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002591476,0.0001780445,0.001079405,0.0001461952,0.0002150479,0.0000151819,0.00007586036,0.00005787261,0.001010363],"category_scores_gemma":[0.0004968601,0.0001154642,0.0001901877,0.00030297,0.0001400023,0.0002168265,0.00005833842,0.0009962419,7.053167e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005147619,"about_ca_system_score_gemma":0.0002498507,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000006374751,"about_ca_topic_score_gemma":4.615837e-7,"domain_scores_codex":[0.99675,0.0012878,0.0009934129,0.0002271194,0.0004915662,0.0002501268],"domain_scores_gemma":[0.9964005,0.001920826,0.001128146,0.0001479633,0.000187887,0.0002146371],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001350773,0.0009759699,0.9769838,0.0004311646,0.0001622772,0.001184256,0.0001648273,0.002489998,0.0003217892,0.000006814254,0.002118582,0.01380973],"study_design_scores_gemma":[0.002466385,0.01455628,0.5913163,0.001665373,0.0009900181,0.03788693,0.0005295672,0.1806469,0.0001624805,0.000005752394,0.1692836,0.0004903985],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9850219,0.008692887,0.00003080224,0.005726246,0.0002528303,0.0002139086,9.523198e-7,0.00001290063,0.00004756539],"genre_scores_gemma":[0.9837909,0.01409238,0.0006148253,0.0013187,0.0001326775,0.00001686297,0.000007688383,0.00001294028,0.00001307244],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3856675,"threshold_uncertainty_score":0.9999028,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06523879130970361,"score_gpt":0.3537599738036798,"score_spread":0.2885211824939762,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}