{"id":"W4402406486","doi":"10.23889/ijpds.v9i5.2867","title":"Creating a Data Cleaning and Pre-Processing Module for Generalisable Data Linkage","year":2024,"lang":"en","type":"article","venue":"International Journal for Population Data Science","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta; University of Calgary; Alberta Health; Alberta Health Services","funders":"","keywords":"Linkage (software); Database; Data processing; Computer science; Process engineering; Engineering; Chemistry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","scholarly_communication","open_science"],"consensus_categories":["scholarly_communication"],"category_scores_codex":[0.01885901,0.0001269114,0.0001631728,0.0004801859,0.0009448366,0.008934693,0.01183317,0.00003492267,0.00004662177],"category_scores_gemma":[0.01163909,0.0001002736,0.00002646041,0.0005364728,0.0001625154,0.02140268,0.006603636,0.0001478098,0.000009650163],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007241251,"about_ca_system_score_gemma":0.0002306656,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002134343,"about_ca_topic_score_gemma":0.0001923253,"domain_scores_codex":[0.99541,0.0000585114,0.0008503473,0.001345507,0.00204644,0.0002891714],"domain_scores_gemma":[0.9959332,0.0008187246,0.0003673956,0.002178613,0.0005591608,0.000142958],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007189693,0.00005082734,0.001696893,0.00005683461,0.00006035331,0.00001008959,0.0003949009,0.001904227,0.0004742901,0.03526304,0.08484333,0.8751733],"study_design_scores_gemma":[0.0001875397,0.00001773553,0.00183617,0.0001132461,0.00002306241,0.00004353309,0.0002244698,0.7818284,0.00001457361,0.01824478,0.1973558,0.0001106921],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0158702,0.0005829029,0.9624825,0.004414216,0.003900197,0.0003831852,0.01204077,0.00005807594,0.0002680311],"genre_scores_gemma":[0.7222004,0.000163695,0.2584679,0.000757964,0.002233456,0.00001165572,0.01425058,0.00002810961,0.00188626],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8750626,"threshold_uncertainty_score":0.9966863,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4833771817173695,"score_gpt":0.5563532414405485,"score_spread":0.07297605972317905,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}