{"id":"W2971290973","doi":"10.14778/3342263.3342633","title":"An intermediate representation for optimizing machine learning pipelines","year":2019,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":48,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Berlin Center for Machine Learning; Banting and Best Diabetes Centre, University of Toronto; York University","keywords":"Computer science; Preprocessor; Pipeline transport; Domain (mathematical analysis); External Data Representation; Representation (politics); Semantics (computer science); Data pre-processing; Feature (linguistics); Programming language; Feature engineering; Artificial intelligence; Theoretical computer science; Machine learning; Deep learning","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002646128,0.001630126,0.0009654404,0.001495069,0.0009587882,0.00444497,0.003571244,0.00134324,0.0106144],"category_scores_gemma":[0.009615853,0.0008600696,0.002298438,0.002019137,0.001200901,0.004608551,0.002981791,0.002878482,0.004170553],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001604296,"about_ca_system_score_gemma":0.002870207,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002789357,"about_ca_topic_score_gemma":0.004125796,"domain_scores_codex":[0.9977126,0.0005064976,0.0002472793,0.0004259828,0.0007855613,0.0003220566],"domain_scores_gemma":[0.9963546,0.001226631,0.0001905527,0.001330586,0.0007781622,0.0001195235],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007729091,0.0003273059,0.002673076,0.0005844867,0.0001057549,0.0003450682,0.0005044057,0.3759045,0.01724781,0.2793814,0.03301986,0.2891334],"study_design_scores_gemma":[0.00006140485,0.0001172864,0.0002832862,0.0000667152,0.00005529455,0.00007428202,0.00009005336,0.8231875,0.02087875,0.1340856,0.02105932,0.00004039722],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.003811928,0.00007099726,0.9846416,0.0001701272,0.00004003182,0.00005224551,0.0004355786,0.00887308,0.001904403],"genre_scores_gemma":[0.1046471,0.0001483032,0.8854604,0.0001856129,0.00006115962,0.0003695266,0.003320123,0.002843917,0.002963874],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0106144,"threshold_uncertainty_score":0.03550869,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08481634541627903,"score_gpt":0.3758174011326433,"score_spread":0.2910010557163643,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}