{"id":"W4400470487","doi":"10.20944/preprints202407.0425.v1","title":"An Unsupervised Error Detection Methodology for Detecting Mislabels in Healthcare Analytics","year":2024,"lang":"en","type":"preprint","venue":"Preprints.org","topic":"Artificial Intelligence in Healthcare","field":"Health Professions","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Interpretability; Computer science; Cluster analysis; Unsupervised learning; Artificial intelligence; Reliability (semiconductor); Data mining; Machine learning; Pattern recognition (psychology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006970004,0.001307804,0.001392888,0.005696539,0.001656321,0.002289895,0.002845464,0.001848318,0.0006756574],"category_scores_gemma":[0.03038861,0.0005625809,0.001473584,0.003743627,0.0012853,0.002244189,0.002126646,0.002320269,0.0007046386],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001187612,"about_ca_system_score_gemma":0.003090513,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003350175,"about_ca_topic_score_gemma":0.005274978,"domain_scores_codex":[0.9906889,0.001984661,0.001177785,0.002743312,0.003077334,0.0003279307],"domain_scores_gemma":[0.9725519,0.01071057,0.004959328,0.004445833,0.006952443,0.0003799594],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004698554,0.0006101211,0.04468965,0.0004222486,0.0004885858,0.0004782113,0.001119084,0.04980147,0.02445103,0.00893723,0.007993714,0.8605388],"study_design_scores_gemma":[0.00006463328,0.0002403592,0.01375524,0.0001303458,0.0001251322,0.0009766397,0.0002948197,0.8977664,0.05807838,0.02030954,0.008135618,0.0001229139],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02023276,0.0002290412,0.9764245,0.000239853,0.00007897497,0.0002498209,0.0002749666,0.001772809,0.0004973008],"genre_scores_gemma":[0.2139367,0.0001823728,0.7823482,0.0002618293,0.0001082115,0.0003825109,0.001007466,0.0002288026,0.001544021],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006970004,"threshold_uncertainty_score":0.03686136,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.705493801331435,"score_gpt":0.6047276982236441,"score_spread":0.1007661031077909,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}