{"id":"W4408808277","doi":"10.6000/1929-6029.2025.14.16","title":"A Choice of Performance Metrics for Evaluating Predictive Accuracy of Survival Models","year":2025,"lang":"en","type":"article","venue":"International Journal of Statistics in Medical Research","topic":"Health Systems, Economic Evaluations, Quality of Life","field":"Economics, Econometrics and Finance","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Statistics; Econometrics; Machine learning; Artificial intelligence; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.05166012,0.002264112,0.002101301,0.008838721,0.001177687,0.004180542,0.001850727,0.002471291,0.001347211],"category_scores_gemma":[0.1631635,0.0004293904,0.002149801,0.006495771,0.001853747,0.00382229,0.002659703,0.003143187,0.0005123932],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00145393,"about_ca_system_score_gemma":0.002691078,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006349529,"about_ca_topic_score_gemma":0.004718349,"domain_scores_codex":[0.9805771,0.01096162,0.001980634,0.001877817,0.003903603,0.0006992749],"domain_scores_gemma":[0.8713137,0.1036609,0.00768606,0.00756296,0.008573095,0.001203319],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0009310615,0.0004633248,0.2135353,0.001253088,0.002082701,0.0004639299,0.0008846045,0.4997881,0.002233451,0.02292454,0.01171473,0.2437252],"study_design_scores_gemma":[0.00005543671,0.0007538648,0.03298618,0.0005052273,0.000297576,0.0004951266,0.0008916223,0.9378392,0.003049347,0.01897135,0.003979818,0.0001753155],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3321431,0.01323174,0.6314524,0.003507798,0.0008713315,0.0006416529,0.005271184,0.002438525,0.01044229],"genre_scores_gemma":[0.8945378,0.001693854,0.09795432,0.0003117512,0.0002568838,0.0002949657,0.003967587,0.0002026763,0.0007801913],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9483399,"threshold_uncertainty_score":0.2732081,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6491725716129512,"score_gpt":0.6344758245705483,"score_spread":0.01469674704240287,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}