{"id":"W2999004052","doi":"10.1609/hcomp.v7i1.5274","title":"A Hybrid Approach to Identifying Unknown Unknowns of Predictive Models","year":2019,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Human Computation and Crowdsourcing","topic":"Mobile Crowdsensing and Crowdsourcing","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Crowdsourcing; Computer science; Task (project management); Machine learning; Set (abstract data type); Artificial intelligence; Data mining; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006632691,0.002531257,0.003249929,0.004575905,0.001986999,0.005271351,0.005353463,0.004033286,0.002343828],"category_scores_gemma":[0.03103063,0.001545794,0.002354284,0.003405782,0.002975093,0.005578817,0.005159665,0.005545251,0.001186417],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001831484,"about_ca_system_score_gemma":0.00269895,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006949726,"about_ca_topic_score_gemma":0.008463151,"domain_scores_codex":[0.99291,0.00250021,0.0004458264,0.00202149,0.001773628,0.0003487235],"domain_scores_gemma":[0.973235,0.01871432,0.001958644,0.003701077,0.00193412,0.0004568896],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006211922,0.0005451645,0.009186961,0.0005581885,0.0006004997,0.0008471843,0.001553748,0.5060061,0.005661767,0.05553769,0.007979921,0.4109017],"study_design_scores_gemma":[0.00002781004,0.00005159332,0.0004198454,0.00005146399,0.00005221513,0.0001436955,0.00009049723,0.9437753,0.0014552,0.05164784,0.002248238,0.0000362841],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01093142,0.0003771367,0.9851692,0.0007948717,0.00004036806,0.0001420761,0.0001956201,0.0009317782,0.001417542],"genre_scores_gemma":[0.4088714,0.0003950439,0.5838196,0.0009624808,0.0003914422,0.0004684638,0.001109429,0.0002128067,0.003769303],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006949726,"threshold_uncertainty_score":0.03507739,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04897807387878888,"score_gpt":0.2707203747712563,"score_spread":0.2217423008924674,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}