{"id":"W3159062657","doi":"","title":"Beyond Marginal Uncertainty: How Accurately can Bayesian Regression Models Estimate Posterior Predictive Correlations?","year":2021,"lang":"en","type":"article","venue":"International Conference on Artificial Intelligence and Statistics","topic":"Machine Learning and Algorithms","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Machine learning; Bayesian probability; Benchmark (surveying); Artificial intelligence; Gaussian process; Regression; Consistency (knowledge bases); Benchmarking; Marginal likelihood; Statistics; Mathematics; Gaussian","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02136753,0.002413437,0.002354017,0.002712895,0.000794051,0.005626956,0.003080008,0.003531274,0.002304031],"category_scores_gemma":[0.1301337,0.001128356,0.001313035,0.002344586,0.002586343,0.0119843,0.004050173,0.00455793,0.001146372],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00180888,"about_ca_system_score_gemma":0.002389542,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008246153,"about_ca_topic_score_gemma":0.006553962,"domain_scores_codex":[0.9914678,0.00440977,0.0004797743,0.001316522,0.001948198,0.0003780279],"domain_scores_gemma":[0.9381552,0.04605464,0.003797666,0.00651987,0.004524191,0.0009483529],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003751885,0.00011723,0.01365021,0.0002725603,0.0003887464,0.00008457748,0.0002539606,0.857094,0.001327779,0.0374912,0.003379015,0.08556554],"study_design_scores_gemma":[0.00001718738,0.00005993045,0.001353204,0.00008616719,0.00003249438,0.00005408418,0.00005322791,0.9447827,0.00177578,0.05084475,0.0008951101,0.00004537502],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06891219,0.001740787,0.919474,0.00218344,0.0001143752,0.0000701321,0.0009751533,0.001861185,0.004668676],"genre_scores_gemma":[0.8381957,0.001085234,0.1559328,0.0006090891,0.000210653,0.0001699247,0.001756671,0.0008082567,0.001231749],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02136753,"threshold_uncertainty_score":0.1130037,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08572084720730128,"score_gpt":0.3550802520154215,"score_spread":0.2693594048081202,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}