{"id":"W4362486594","doi":"10.1117/12.2654419","title":"Using uncertainty quantification to improve reliability of video-based skill assessment metrics in central venous catheterization","year":2023,"lang":"en","type":"article","venue":"","topic":"Hemodynamic Monitoring and Therapy","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Ground truth; Computer science; Metric (unit); Bounding overwatch; Reliability (semiconductor); Receiver operating characteristic; Minimum bounding box; Path (computing); False positive paradox; Data mining; Artificial intelligence; Statistics; Mathematics; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0066425,0.001329784,0.0008780265,0.002700576,0.0004075787,0.001557467,0.001044874,0.001083124,0.0007886728],"category_scores_gemma":[0.040857,0.0004857951,0.000466423,0.0009989657,0.0005765862,0.001669508,0.001947268,0.000935343,0.0003108555],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009500535,"about_ca_system_score_gemma":0.001207719,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005204559,"about_ca_topic_score_gemma":0.004136971,"domain_scores_codex":[0.9947278,0.001916273,0.0003964271,0.001012069,0.001687058,0.0002604105],"domain_scores_gemma":[0.9753208,0.0164878,0.002178365,0.001389764,0.004278579,0.0003446444],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0009008912,0.0002392967,0.07238119,0.0004524273,0.0003121866,0.0002351221,0.0008220974,0.2166148,0.03698178,0.001746237,0.002246295,0.6670676],"study_design_scores_gemma":[0.00001483964,0.0002895365,0.02267913,0.00008296056,0.00006415675,0.000188008,0.00009543614,0.9409707,0.03231034,0.001874401,0.001354674,0.00007591024],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2765356,0.001604151,0.7160621,0.0002797049,0.0001402761,0.0001275155,0.0003145362,0.002827488,0.00210865],"genre_scores_gemma":[0.899119,0.0002328163,0.09950027,0.0000722476,0.00005118209,0.00006991001,0.0003360202,0.0001312093,0.0004872225],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0066425,"threshold_uncertainty_score":0.03512937,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04241291746821341,"score_gpt":0.3658741586069827,"score_spread":0.3234612411387693,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}