{"id":"W4318148225","doi":"10.1109/bigdata55660.2022.10020386","title":"Adaptive Method for Machine Learning Model Selection in Data Science Projects","year":2022,"lang":"en","type":"article","venue":"2022 IEEE International Conference on Big Data (Big Data)","topic":"Software Engineering Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Selection (genetic algorithm); Machine learning; Process (computing); Heuristics; Feature selection; Model selection; Artificial intelligence; Data mining","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01356869,0.00109503,0.0009174866,0.002928711,0.0009752389,0.001830918,0.002637245,0.001639127,0.00381531],"category_scores_gemma":[0.04898661,0.0007623716,0.001702372,0.002277857,0.0009944342,0.002137985,0.002343253,0.003348596,0.0006640197],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001568343,"about_ca_system_score_gemma":0.00286122,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004615558,"about_ca_topic_score_gemma":0.007426095,"domain_scores_codex":[0.9905923,0.006097058,0.0004585738,0.001169952,0.001412179,0.0002699872],"domain_scores_gemma":[0.9591773,0.03396612,0.001570697,0.002282236,0.002600614,0.0004029306],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000418662,0.0003794291,0.01200561,0.0003297956,0.0004161185,0.0004100631,0.0008298467,0.5824137,0.003129953,0.06175525,0.004448804,0.3334629],"study_design_scores_gemma":[0.00002606363,0.00003024467,0.000264733,0.0000160844,0.00001652133,0.00002964246,0.00004130354,0.983095,0.0006266172,0.01473688,0.001105469,0.00001143177],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.005203799,0.00004400319,0.9935132,0.0001016857,0.0000131216,0.0001056363,0.00004133253,0.0005492142,0.000428109],"genre_scores_gemma":[0.1561818,0.00006282736,0.8414322,0.0001569201,0.00003315354,0.000732776,0.0002989469,0.0002084426,0.0008929871],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01356869,"threshold_uncertainty_score":0.07175893,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4726800557815008,"score_gpt":0.4217361885031067,"score_spread":0.05094386727839412,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}