{"id":"W4402475639","doi":"10.1109/ccece59415.2024.10667232","title":"Are Large Language Models General-Purpose Solvers for Dialogue Breakdown Detection? An Empirical Investigation","year":2024,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Empirical research; Natural language processing; Epistemology; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02986024,0.002666614,0.001752111,0.001780479,0.001072515,0.004666631,0.003168889,0.002557084,0.007296931],"category_scores_gemma":[0.1780185,0.001018435,0.001450309,0.001714766,0.001785818,0.01085048,0.00284765,0.006056342,0.002852133],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00194166,"about_ca_system_score_gemma":0.002277176,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006361765,"about_ca_topic_score_gemma":0.007843548,"domain_scores_codex":[0.9761097,0.01651104,0.0007905666,0.004276227,0.001610106,0.0007023687],"domain_scores_gemma":[0.7861999,0.1915493,0.004239373,0.01121414,0.004751315,0.002045928],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00472718,0.002857421,0.1350061,0.004132782,0.001497005,0.0008965047,0.006765279,0.2717852,0.00472308,0.02734004,0.04467182,0.4955975],"study_design_scores_gemma":[0.0002158476,0.0003502594,0.007713696,0.0002306978,0.0001879321,0.0003109374,0.001459543,0.9595755,0.001098491,0.02355456,0.005239638,0.00006290576],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6614364,0.008733458,0.2880306,0.009009509,0.0006546177,0.001054142,0.005004548,0.006607579,0.01946916],"genre_scores_gemma":[0.9417062,0.000685505,0.05077859,0.0008258032,0.0001678091,0.0003480656,0.003718655,0.0004920583,0.001277273],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02986024,"threshold_uncertainty_score":0.157918,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04853789792381037,"score_gpt":0.307442302641208,"score_spread":0.2589044047173977,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}