{"id":"W4415421841","doi":"10.15439/2025f5708","title":"Out-Of-Distribution Is Not Magic: The Clash Between Rejection Rate and Model Success","year":2025,"lang":"en","type":"article","venue":"Annals of Computer Science and Information Systems","topic":"Data Stream Mining Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Government (linguistics); Perspective (graphical)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.08799902,0.001665765,0.002740063,0.002117557,0.001371609,0.004922363,0.003034867,0.003109959,0.001256977],"category_scores_gemma":[0.2605357,0.0005842992,0.001654395,0.001179735,0.003987353,0.006834583,0.003435386,0.006077338,0.000571588],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002045268,"about_ca_system_score_gemma":0.002302897,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004188403,"about_ca_topic_score_gemma":0.00244389,"domain_scores_codex":[0.9568265,0.02726425,0.001870077,0.005561089,0.007308819,0.001169336],"domain_scores_gemma":[0.6675904,0.2814782,0.01135806,0.02599063,0.01169571,0.001886915],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003458355,0.0006898047,0.1547089,0.001180006,0.002506972,0.001056108,0.00176091,0.3665121,0.006553653,0.1112483,0.01156227,0.3387626],"study_design_scores_gemma":[0.0001113537,0.0005035404,0.01085113,0.0002267384,0.0002518897,0.0005002792,0.0004451955,0.8802952,0.005587487,0.09686765,0.004227528,0.0001318976],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1399772,0.003896685,0.8409572,0.00767634,0.0004807441,0.0001831082,0.0002897075,0.001382495,0.00515648],"genre_scores_gemma":[0.9234179,0.0008669389,0.07176235,0.00145858,0.0003448759,0.0001956017,0.0004167534,0.0004406802,0.001096261],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.08799902,"threshold_uncertainty_score":0.4653888,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05066555381671407,"score_gpt":0.3211341940343513,"score_spread":0.2704686402176373,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}