{"id":"W4417041459","doi":"10.48550/arxiv.2504.20006","title":"Chatbot Arena Meets Nuggets: Towards Explanations and Diagnostics in the Evaluation of LLM Responses","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"AI in Service Interactions","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Chatbot; Context (archaeology); Quality (philosophy); Code (set theory); Work (physics)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0126633,0.001605884,0.0009496563,0.006760226,0.0006480529,0.003623245,0.00173494,0.002592987,0.005437031],"category_scores_gemma":[0.07080977,0.0004903026,0.0007320713,0.002114737,0.001632288,0.003800585,0.00316371,0.002215188,0.001740046],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001325419,"about_ca_system_score_gemma":0.001179209,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003337077,"about_ca_topic_score_gemma":0.004644669,"domain_scores_codex":[0.9837177,0.01139257,0.0006970824,0.001656077,0.002108552,0.0004279387],"domain_scores_gemma":[0.9282951,0.05924894,0.004232508,0.004157285,0.003037729,0.001028453],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003718471,0.000823342,0.07561862,0.003368961,0.0005075547,0.000785098,0.0159429,0.03806149,0.05300761,0.02116478,0.01867549,0.7683257],"study_design_scores_gemma":[0.0002950219,0.001022321,0.07210267,0.0003511215,0.0001516122,0.0006333428,0.004716584,0.8171006,0.03944182,0.04619434,0.01773936,0.0002512139],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.368142,0.002154717,0.5900308,0.001268229,0.0001301189,0.0008644381,0.004098382,0.02753719,0.00577419],"genre_scores_gemma":[0.7721784,0.0001700745,0.2209712,0.0001946636,0.00005732537,0.0004746469,0.003707196,0.0007803176,0.001466235],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0126633,"threshold_uncertainty_score":0.06697077,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1453284930399543,"score_gpt":0.3884572676797482,"score_spread":0.2431287746397939,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}