{"id":"W4312989454","doi":"10.14778/3554821.3554864","title":"CERTEM","year":2022,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Counterfactual thinking; Computer science; Debugging; Order (exchange); Matching (statistics); State (computer science); Artificial intelligence; Programming language; Epistemology; Mathematics; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003383582,0.00008075943,0.0001556302,0.0001118153,0.0003313689,0.000101898,0.002131376,0.00001009337,0.00083878],"category_scores_gemma":[0.0004871552,0.00004940535,0.0001251153,0.0006803476,0.00006181779,0.0001904094,0.002684553,0.0001177154,0.00005504677],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007334291,"about_ca_system_score_gemma":0.00001699991,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004756431,"about_ca_topic_score_gemma":0.00000229016,"domain_scores_codex":[0.9970824,0.00002790919,0.0004365778,0.0002919813,0.001983293,0.000177852],"domain_scores_gemma":[0.9991055,0.0001100395,0.0003467173,0.0002772029,0.0001190289,0.0000415321],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00005842769,0.0003413858,0.003710906,0.00002570446,0.00004996509,6.496319e-7,0.001925458,0.0001077913,0.006543842,0.3064453,0.6522521,0.02853845],"study_design_scores_gemma":[0.0003338749,0.00009114567,0.003441245,0.000006054944,0.00001788654,0.000005499219,0.007972308,0.0001709156,0.008237335,0.09827807,0.8813429,0.0001027961],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5380926,0.0002681255,0.0002071044,0.04527339,0.002831856,0.001761176,0.0002057389,0.0001297142,0.4112304],"genre_scores_gemma":[0.9873621,0.000005397933,0.0002385133,0.0009930928,0.00003097232,0.00008245711,9.26245e-7,0.000005511134,0.01128096],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4492696,"threshold_uncertainty_score":0.9184052,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1511165069730517,"score_gpt":0.3633486292850461,"score_spread":0.2122321223119945,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}