{"id":"W6969072965","doi":"10.5281/zenodo.8344723","title":"Risk-driven Online Testing and Test Case Diversity Analysis for ML-enabled Critical Systems (Replication Package)","year":2023,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Logic, programming, and type systems","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Test script; Python (programming language); Replication (statistics); Test (biology); Cluster analysis; White-box testing; Test data; Function (biology); Control (management)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006423706,0.002108098,0.001127583,0.002191383,0.0004869851,0.00242683,0.003441983,0.001038664,0.1478204],"category_scores_gemma":[0.0303185,0.001520997,0.001781486,0.001641838,0.0004243511,0.002349147,0.002629156,0.001975206,0.07437906],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001034921,"about_ca_system_score_gemma":0.002340594,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003403717,"about_ca_topic_score_gemma":0.002568539,"domain_scores_codex":[0.9963977,0.000825424,0.0003010493,0.0005473524,0.00173352,0.0001948698],"domain_scores_gemma":[0.9826971,0.006221284,0.0007872247,0.005054992,0.004910948,0.000328465],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0008086648,0.0004170372,0.006214531,0.00146591,0.0002564822,0.0003940439,0.000411457,0.04165567,0.00602324,0.0114301,0.6671229,0.2637999],"study_design_scores_gemma":[0.001067034,0.0009999166,0.01220627,0.00107305,0.0002742035,0.00123616,0.0002794284,0.4228584,0.03862165,0.03943985,0.4815202,0.0004237339],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"dataset","genre_scores_codex":[0.006424622,0.0002811123,0.4714473,0.0005119925,0.0002879299,0.001162212,0.07299522,0.4248676,0.02202199],"genre_scores_gemma":[0.09586029,0.0006873424,0.5130324,0.0005687877,0.0002791083,0.004663458,0.2203151,0.132345,0.03224852],"genre_candidate":"dataset","genre_consensus":null,"teacher_disagreement_score":0.1478204,"threshold_uncertainty_score":0.4945086,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06629110077810187,"score_gpt":0.2763247190028231,"score_spread":0.2100336182247212,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}