{"id":"W4200001896","doi":"10.3233/faia210316","title":"Data-Centric Machine Learning: Improving Model Performance and Understanding Through Dataset Analysis","year":2021,"lang":"en","type":"book-chapter","venue":"Frontiers in artificial intelligence and applications","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; Research Unit on Children's Psychosocial Maladjustment","funders":"","keywords":"Computer science; Machine learning; Artificial intelligence; Classifier (UML); Training set; Metric (unit); Performance metric; Test set; Test data; Process (computing); Set (abstract data type); Data set; Data mining; Domain (mathematical analysis); Labeled data; Engineering; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008303013,0.001569026,0.001434856,0.002551206,0.0004556241,0.0049906,0.002675147,0.001413733,0.002245452],"category_scores_gemma":[0.03739692,0.0006475396,0.00107619,0.003673142,0.0009339062,0.008096805,0.002211616,0.004055913,0.002134232],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001553683,"about_ca_system_score_gemma":0.001234373,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001441252,"about_ca_topic_score_gemma":0.001946535,"domain_scores_codex":[0.9959307,0.001805597,0.0002369547,0.0007694997,0.001166333,0.00009081681],"domain_scores_gemma":[0.9750405,0.01638394,0.0009910743,0.005416875,0.00199549,0.0001719891],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000182812,0.0002956006,0.01065577,0.001071269,0.0002578375,0.0001712279,0.0008121775,0.09583379,0.01133516,0.0625578,0.04827496,0.7685515],"study_design_scores_gemma":[0.00002290132,0.0001384218,0.004615359,0.0003097005,0.00007209522,0.0002439166,0.0002749887,0.7695078,0.01836636,0.1730501,0.03331838,0.00008001903],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01859375,0.003178415,0.9642522,0.001793381,0.0002025521,0.000168094,0.002490722,0.003286142,0.006034748],"genre_scores_gemma":[0.1433921,0.002756442,0.8366057,0.0007609578,0.0002332152,0.000546826,0.01022668,0.001814339,0.003663882],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008303013,"threshold_uncertainty_score":0.0439111,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1220128723694561,"score_gpt":0.3056667902057088,"score_spread":0.1836539178362527,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}