{"id":"W2513912098","doi":"10.1016/j.shpsa.2016.08.002","title":"How we load our data sets with theories and why we do so purposefully","year":2016,"lang":"en","type":"article","venue":"Studies in History and Philosophy of Science Part A","topic":"Philosophy and History of Science","field":"Arts and Humanities","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université Laval","funders":"","keywords":"Imputation (statistics); Computer science; Perception; Epistemology; Phenomenon; Data science; Philosophy of science; Quality (philosophy); Missing data; Machine learning; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["sts"],"consensus_categories":[],"category_scores_codex":[0.03512235,0.002437118,0.001996077,0.006747997,0.003465732,0.01879152,0.004215088,0.003018191,0.01377102],"category_scores_gemma":[0.3566783,0.002187204,0.003025655,0.007664688,0.005395974,0.02509072,0.007648534,0.01077147,0.01246465],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002837725,"about_ca_system_score_gemma":0.005103838,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01456694,"about_ca_topic_score_gemma":0.01220204,"domain_scores_codex":[0.9691827,0.0156658,0.00232598,0.004107775,0.007797321,0.000920506],"domain_scores_gemma":[0.8046854,0.1162816,0.005176531,0.0474998,0.02332317,0.003033591],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0009964462,0.001153422,0.03438612,0.00142954,0.001273641,0.0003265905,0.008315385,0.02589665,0.01085644,0.05899486,0.1825898,0.6737811],"study_design_scores_gemma":[0.000706306,0.0003702597,0.01831865,0.001302692,0.0006731606,0.0003882842,0.009474373,0.3297862,0.02621881,0.4445395,0.1677465,0.0004753082],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07896404,0.001477381,0.7996275,0.04330254,0.004201986,0.002485177,0.01314912,0.03778209,0.01901016],"genre_scores_gemma":[0.1651512,0.000701106,0.8025418,0.005582901,0.001348268,0.001879882,0.009855792,0.00664552,0.006293523],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9965343,"threshold_uncertainty_score":0.185747,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1571542823427685,"score_gpt":0.2841780619786097,"score_spread":0.1270237796358411,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}