{"id":"W4387532544","doi":"10.1098/rsos.230568","title":"Reducing bias in secondary data analysis via an Explore and Confirm Analysis Workflow (ECAW): a proposal and survey of observational researchers","year":2023,"lang":"en","type":"article","venue":"Royal Society Open Science","topic":"Health, Environment, Cognitive Aging","field":"Environmental Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Medical Research Council; Canadian Institutes of Health Research; Arnold Ventures","keywords":"Observational study; Workflow; Data science; Computer science; Data mining; Database; Statistics; Mathematics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.7271891,0.00190394,0.001452418,0.008511142,0.006965867,0.01338916,0.008325749,0.01061829,0.003669325],"category_scores_gemma":[0.6516061,0.00300913,0.003941434,0.006530446,0.01529893,0.0229277,0.01558565,0.01264082,0.002447527],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.009476488,"about_ca_system_score_gemma":0.07905974,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006376703,"about_ca_topic_score_gemma":0.0059583,"domain_scores_codex":[0.3734752,0.4951743,0.06364377,0.01039742,0.05117315,0.006136065],"domain_scores_gemma":[0.2416648,0.4926633,0.02719099,0.06442847,0.1564819,0.01757052],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.0009008456,0.002253275,0.04116341,0.01024156,0.0002962483,0.0009883922,0.05428876,0.002366938,0.003789841,0.1013526,0.07731284,0.7050453],"study_design_scores_gemma":[0.002514281,0.00795799,0.04016967,0.03976093,0.0008928137,0.00417309,0.05584492,0.04689078,0.007891913,0.143299,0.6492402,0.001364455],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01657372,0.00220986,0.7368999,0.1675376,0.002821957,0.06227851,0.0004665494,0.002521726,0.008690191],"genre_scores_gemma":[0.05371785,0.00260767,0.8770278,0.01624676,0.0008169804,0.04704252,0.0003296124,0.0003431834,0.001867663],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.2728109,"threshold_uncertainty_score":0.3364245,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3171527591360382,"score_gpt":0.4148605485946184,"score_spread":0.09770778945858016,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}