{"id":"W2513912098","doi":"10.1016/j.shpsa.2016.08.002","title":"How we load our data sets with theories and why we do so purposefully","year":2016,"lang":"en","type":"article","venue":"Studies in History and Philosophy of Science Part A","topic":"Philosophy and History of Science","field":"Arts and Humanities","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université Laval","funders":"","keywords":"Imputation (statistics); Computer science; Perception; Epistemology; Phenomenon; Data science; Philosophy of science; Quality (philosophy); Missing data; Machine learning; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["sts"],"consensus_categories":[],"category_scores_codex":[0.001001128,0.0002339504,0.0003740017,0.0002390846,0.001136299,0.00008849674,0.0007396743,0.0000304692,0.00002506157],"category_scores_gemma":[0.0001607216,0.0001435252,0.00002502947,0.0001200275,0.03022962,0.00210364,0.0004204367,0.0001150239,0.000003723781],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001900846,"about_ca_system_score_gemma":0.0002573818,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002018045,"about_ca_topic_score_gemma":0.000277748,"domain_scores_codex":[0.9980624,0.00004593102,0.0002377064,0.0007336732,0.0005814524,0.000338848],"domain_scores_gemma":[0.9988108,0.0001095973,0.0001878436,0.0005388439,0.0002336478,0.0001193087],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001529149,0.00008637274,0.000278363,0.0002366433,0.00003689725,0.00001926835,0.1185547,2.214475e-7,0.0003699646,0.8601418,0.01171156,0.00841135],"study_design_scores_gemma":[0.0004435999,0.0003858465,0.00002743254,0.0005943331,0.00002387255,0.00001126587,0.01022679,0.000002943086,0.00003940378,0.2581349,0.7298091,0.0003004819],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"review","genre_gemma":"empirical","genre_scores_codex":[0.1434643,0.3759204,0.00002636984,0.2389159,0.009053236,0.001347074,0.000655998,0.0002208727,0.2303959],"genre_scores_gemma":[0.9942787,0.003760001,0.0001052348,0.0002830147,0.0003444217,0.00001737726,5.78237e-7,0.000008443634,0.001202197],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8508145,"threshold_uncertainty_score":0.9724095,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1571542823427685,"score_gpt":0.2841780619786097,"score_spread":0.1270237796358411,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}