{"id":"W4312989454","doi":"10.14778/3554821.3554864","title":"CERTEM","year":2022,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Counterfactual thinking; Computer science; Debugging; Order (exchange); Matching (statistics); State (computer science); Artificial intelligence; Programming language; Epistemology; Mathematics; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002817235,0.0015494,0.000869328,0.002934924,0.0008209359,0.003193835,0.004591544,0.00193911,0.0623842],"category_scores_gemma":[0.01598027,0.0006948407,0.001701791,0.002386132,0.0007594683,0.005256001,0.00436355,0.002084599,0.03243056],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001443987,"about_ca_system_score_gemma":0.002684813,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005388842,"about_ca_topic_score_gemma":0.008337027,"domain_scores_codex":[0.9977432,0.0003675203,0.0002063163,0.0007727333,0.0007160146,0.0001941962],"domain_scores_gemma":[0.9949058,0.001396849,0.0002995175,0.002214908,0.001025225,0.0001577395],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0005602087,0.000191936,0.004660751,0.001781411,0.0002013459,0.0004204731,0.0003416282,0.01044522,0.003040561,0.04361901,0.7265651,0.2081724],"study_design_scores_gemma":[0.0002302288,0.0001238647,0.002483638,0.0002465304,0.00007613799,0.0007067747,0.0001986834,0.119076,0.01260409,0.04555174,0.8186055,0.00009685053],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"software","genre_gemma":"other","genre_scores_codex":[0.01217798,0.002261776,0.2782329,0.003408116,0.0008793392,0.0008359707,0.118653,0.523381,0.06016994],"genre_scores_gemma":[0.09316325,0.001429756,0.3643762,0.002379358,0.0002598063,0.0008380946,0.4693309,0.03258862,0.03563396],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.9376158,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1511165069730517,"score_gpt":0.3633486292850461,"score_spread":0.2122321223119945,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}