{"id":"W4394998074","doi":"10.21203/rs.3.rs-4266384/v1","title":"Assessing Model Validity Using KL Divergence and Beta-Stacy Process for Right Censored Data","year":2024,"lang":"en","type":"preprint","venue":"Research Square","topic":"Statistical Methods and Inference","field":"Mathematics","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Divergence (linguistics); Process (computing); BETA (programming language); Econometrics; Statistics; Computer science; Mathematics; Programming language; Philosophy; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.005050102,0.0003308022,0.0006112278,0.0002273689,0.0004641343,0.0009474898,0.001071719,0.0003370479,0.00008710706],"category_scores_gemma":[0.01075939,0.0002694418,0.00008482554,0.0002801878,0.0003048677,0.0002361356,0.005568194,0.00164363,0.000005404389],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001507408,"about_ca_system_score_gemma":0.0009514618,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00008446175,"about_ca_topic_score_gemma":0.00001247123,"domain_scores_codex":[0.9956685,0.0005962741,0.0005227195,0.001300556,0.001168151,0.0007437726],"domain_scores_gemma":[0.9926555,0.004429709,0.000141441,0.001452197,0.00105939,0.0002618072],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002245364,0.0008364382,0.001572862,0.1484035,0.0005507632,0.0002055537,0.003475122,0.001986921,0.001638703,0.7896085,0.01128828,0.04020886],"study_design_scores_gemma":[0.00007163285,0.00002303003,0.00004774821,0.001242692,0.00007221443,0.000001897557,0.0001400403,0.4799193,0.0002917213,0.5179605,0.00005087264,0.0001782861],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1154554,0.0005517958,0.8748921,0.0002915759,0.000273359,0.001847666,0.005643,0.0001169055,0.0009282457],"genre_scores_gemma":[0.3617788,0.0001181188,0.6372575,0.000007532441,0.0002571208,0.0001539136,0.0001867068,0.00007937779,0.0001610285],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.4779324,"threshold_uncertainty_score":0.9999758,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7761963905194896,"score_gpt":0.6333939167535847,"score_spread":0.1428024737659049,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}