{"id":"W4409168760","doi":"10.1145/3728373","title":"Recall, Robustness, and Lexicographic Evaluation","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Recommender Systems","topic":"Topic Modeling","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Microsoft (Canada)","funders":"","keywords":"Lexicographical order; Robustness (evolution); Recall; Computer science; Artificial intelligence; Cognitive psychology; Mathematics; Psychology; Chemistry; Combinatorics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0535368,0.001497951,0.001920506,0.009679002,0.001494968,0.005927205,0.001565361,0.001962311,0.003206602],"category_scores_gemma":[0.2746973,0.0005074571,0.001683144,0.008139462,0.007057668,0.01188412,0.004184558,0.002211819,0.0006979393],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003058705,"about_ca_system_score_gemma":0.001593243,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001653057,"about_ca_topic_score_gemma":0.001328455,"domain_scores_codex":[0.9248484,0.04382844,0.006632861,0.006192614,0.01727263,0.001225051],"domain_scores_gemma":[0.6673087,0.2570056,0.02784874,0.03025647,0.01569324,0.001887259],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.002077658,0.0005038314,0.09188061,0.002097065,0.001356691,0.0003669215,0.004806949,0.05743133,0.004201259,0.4400789,0.006346952,0.3888517],"study_design_scores_gemma":[0.0002313953,0.002082163,0.0431693,0.0009665648,0.0005911299,0.001039032,0.002119703,0.1389922,0.01099988,0.780461,0.01878065,0.0005670422],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1649524,0.005406906,0.7915114,0.002563534,0.0002845433,0.0006626607,0.00144892,0.0006161478,0.03255344],"genre_scores_gemma":[0.840283,0.0008936907,0.1538224,0.0006024524,0.00041086,0.0006187239,0.001176761,0.0001870291,0.002005007],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0535368,"threshold_uncertainty_score":0.283133,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05255388057789485,"score_gpt":0.2979719475362509,"score_spread":0.2454180669583561,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}