{"id":"W4379469928","doi":"10.48550/arxiv.2306.01196","title":"An Effective Meaningful Way to Evaluate Survival Models","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Statistical Methods and Inference","field":"Mathematics","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"National Institute of Neurological Disorders and Stroke; National Institute of Diabetes and Digestive and Kidney Diseases; National Heart, Lung, and Blood Institute; Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute; National Institutes of Health; National Science Foundation","keywords":"Metric (unit); Event (particle physics); Computer science; Set (abstract data type); Data mining; Machine learning; Statistics; Rank (graph theory); Artificial intelligence; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02942152,0.002195938,0.001741753,0.00579932,0.0007471966,0.003305206,0.001733632,0.002230528,0.002844036],"category_scores_gemma":[0.1433479,0.00044927,0.001138087,0.003585793,0.001867588,0.004279592,0.002390679,0.002991882,0.001079323],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001568741,"about_ca_system_score_gemma":0.001611788,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002301078,"about_ca_topic_score_gemma":0.002652917,"domain_scores_codex":[0.9835861,0.009677202,0.001606522,0.001624621,0.003235386,0.0002703398],"domain_scores_gemma":[0.90005,0.07006042,0.007534333,0.01266973,0.008505092,0.001180496],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0005114365,0.0004929404,0.06288181,0.001356248,0.001730616,0.0002209878,0.0005656162,0.5360163,0.005539462,0.1238904,0.03352535,0.2332688],"study_design_scores_gemma":[0.00006524164,0.0009198147,0.01194737,0.0002611628,0.0001157712,0.0003756202,0.0003581455,0.8556256,0.005128394,0.109467,0.01558925,0.0001466582],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.04549874,0.002277888,0.9374609,0.001976124,0.0004876214,0.000327883,0.006621872,0.001764971,0.00358404],"genre_scores_gemma":[0.5568317,0.00114006,0.4257693,0.0008953448,0.0004790179,0.0007062364,0.0124622,0.0004148012,0.001301424],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02942152,"threshold_uncertainty_score":0.1555977,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3588678959718906,"score_gpt":0.3261002826964549,"score_spread":0.03276761327543565,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}