{"id":"W4399615587","doi":"10.1007/s10664-024-10450-y","title":"How far are we with automated machine learning? characterization and challenges of AutoML toolkits","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"ca_institutions":"York University; University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Characterization (materials science); Computer science; Artificial intelligence; Machine learning; Nanotechnology; Materials science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06015005,0.001028521,0.001817582,0.0063737,0.003567042,0.02285809,0.007983534,0.005934447,0.01066951],"category_scores_gemma":[0.164445,0.001829186,0.001360554,0.005276809,0.01530114,0.05965002,0.01503355,0.01106266,0.008584305],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003980747,"about_ca_system_score_gemma":0.008138442,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004488589,"about_ca_topic_score_gemma":0.004068317,"domain_scores_codex":[0.93855,0.03318175,0.003430624,0.005697851,0.015436,0.003703799],"domain_scores_gemma":[0.73447,0.1441351,0.007687919,0.07346199,0.0312546,0.008990596],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","study_design_scores_codex":[0.0003514899,0.0003387429,0.02034183,0.0006984515,0.00007919706,0.0002130211,0.004657913,0.005330692,0.002012727,0.5948865,0.02885313,0.3422363],"study_design_scores_gemma":[0.00005887493,0.000117442,0.005944434,0.001011871,0.00003622854,0.000702321,0.0051488,0.0545691,0.00378072,0.7864885,0.1419605,0.0001811317],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1299201,0.01588243,0.5534344,0.2262855,0.0007308333,0.0002673959,0.001716764,0.01226762,0.05949492],"genre_scores_gemma":[0.5914972,0.005695863,0.3721414,0.009213519,0.0009189151,0.0005719244,0.002418266,0.006259265,0.01128368],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.93985,"threshold_uncertainty_score":0.3181076,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02592931027558718,"score_gpt":0.2455855985960089,"score_spread":0.2196562883204217,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}