{"id":"W3083318759","doi":"10.1109/tvcg.2020.3030462","title":"Table Scraps: An Actionable Framework for Multi-Table Data Wrangling From An Artifact Study of Computational Journalism","year":2020,"lang":"en","type":"preprint","venue":"IEEE Transactions on Visualization and Computer Graphics","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Parallels; Artifact (error); Table (database); Context (archaeology); Journalism; Data science; Data mining; Artificial intelligence; Engineering; Political science; Geography","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01478771,0.001600933,0.0009864809,0.01044247,0.004898828,0.01252539,0.005027147,0.002786398,0.01038362],"category_scores_gemma":[0.03041966,0.001852045,0.00497816,0.006947801,0.01351745,0.01915243,0.01013304,0.004708907,0.002430897],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002989237,"about_ca_system_score_gemma":0.004353645,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00996618,"about_ca_topic_score_gemma":0.01379894,"domain_scores_codex":[0.9870908,0.006577163,0.001462433,0.002149464,0.002054625,0.0006654698],"domain_scores_gemma":[0.967626,0.01566284,0.002677612,0.0106043,0.002093999,0.001335351],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00004839512,0.00005695165,0.002503492,0.00028095,0.00003257608,0.0004001833,0.01355393,0.003546908,0.001258171,0.9270211,0.004473069,0.04682442],"study_design_scores_gemma":[0.00003170221,0.00007772959,0.001508392,0.0004639014,0.00006256276,0.0007811314,0.006712772,0.0358967,0.002617188,0.7514079,0.2003205,0.000119529],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004224915,0.0002345502,0.9852092,0.001003801,0.0000792461,0.0003018309,0.0005414724,0.002119933,0.006285075],"genre_scores_gemma":[0.05587696,0.0002216837,0.9382107,0.0002179629,0.0000644656,0.0005815393,0.001079635,0.0007650268,0.002982062],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01478771,"threshold_uncertainty_score":0.07820576,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.335592819321051,"score_gpt":0.4663434896748334,"score_spread":0.1307506703537824,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}