{"id":"W3168854329","doi":"10.14778/3467861.3467872","title":"Data acquisition for improving machine learning models","year":2021,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":37,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; York University","funders":"","keywords":"Data acquisition; Process (computing); Data modeling; Online machine learning; Knowledge acquisition; Training set; Annotation; Data integration; Supervised learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01217015,0.001984652,0.001995232,0.002064049,0.001000627,0.002637414,0.003144487,0.002260782,0.004829378],"category_scores_gemma":[0.07221156,0.001099433,0.001831326,0.002853233,0.001716999,0.007616462,0.004229279,0.00529235,0.001794301],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001708925,"about_ca_system_score_gemma":0.002996522,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004274661,"about_ca_topic_score_gemma":0.004387464,"domain_scores_codex":[0.9923843,0.003929856,0.0004917698,0.001188477,0.001700982,0.0003047636],"domain_scores_gemma":[0.9549772,0.03214775,0.001453923,0.008248189,0.002806457,0.0003666695],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001120968,0.001002643,0.0115514,0.0004596727,0.0002915233,0.0001994386,0.0005525046,0.4463065,0.007207052,0.04957299,0.009138313,0.4725969],"study_design_scores_gemma":[0.00006695604,0.0001516872,0.0006015818,0.000036101,0.00003846759,0.00004951004,0.0000795961,0.9666936,0.005030031,0.02424877,0.002983297,0.0000203762],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.04304781,0.001238979,0.947251,0.001810658,0.00009071598,0.000246963,0.0006264637,0.003211455,0.002475875],"genre_scores_gemma":[0.4236725,0.0007829602,0.569904,0.000881276,0.0002121537,0.0005418159,0.002167728,0.0004238983,0.001413668],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01217015,"threshold_uncertainty_score":0.0643627,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2848510771576441,"score_gpt":0.3865902747486838,"score_spread":0.1017391975910397,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}