{"id":"W6944970400","doi":"10.2312/evs.20231036","title":"RiskFix: Supporting Expert Validation of Predictive Timeseries Models in High-Intensity Settings","year":2023,"lang":"en","type":"article","venue":"Open MIND","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Hospital for Sick Children; University of Toronto","funders":"","keywords":"Workflow; Domain (mathematical analysis); Subject-matter expert; Bottleneck; Software; Domain knowledge; Ground truth","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007197051,0.00007101516,0.0002285287,0.0001236276,0.00005354151,0.00002041331,0.00009467219,0.00007429262,0.0004393647],"category_scores_gemma":[0.0003773116,0.00006734896,0.0000266597,0.0003574155,0.00004612812,0.0003150142,0.00007688044,0.00011123,0.0001439808],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006484437,"about_ca_system_score_gemma":0.0001963205,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003826266,"about_ca_topic_score_gemma":0.00007264235,"domain_scores_codex":[0.9989697,0.00003568893,0.0004370859,0.0002146159,0.0001544248,0.0001884575],"domain_scores_gemma":[0.9992922,0.00009492099,0.0001510555,0.0001729964,0.0002283116,0.00006049114],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00213125,0.0005705804,0.2000751,0.0002091877,0.00009966641,0.00007965419,0.2460676,0.004428471,0.02379359,0.0001357029,0.01521477,0.5071944],"study_design_scores_gemma":[0.00015426,0.00041669,0.01692118,0.0004547061,0.00003231645,0.00001187387,0.0604423,0.02938354,0.8869405,0.004264466,0.0008150284,0.0001630851],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9943447,0.00001327965,0.0001578643,0.003045241,0.0001929727,0.0005207776,0.000009941314,0.000005382762,0.001709859],"genre_scores_gemma":[0.9974321,0.0000268381,0.001570805,0.00008682365,0.00009083961,0.00002929518,0.0001340686,0.000009451211,0.0006197399],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.863147,"threshold_uncertainty_score":0.5784195,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2161478756960769,"score_gpt":0.455522485278761,"score_spread":0.2393746095826841,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}