{"id":"W7125908856","doi":"10.1109/ase63991.2025.00067","title":"Watson: A Cognitive Observability Framework for the Reasoning of LLM-Powered Agents","year":2025,"lang":"","type":"article","venue":"","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University; Huawei Technologies (Canada)","funders":"","keywords":"Observability; Debugging; Transparency (behavior); Software; Benchmark (surveying); Semantic reasoner; Autonomous agent; Cognition; Model-based reasoning; Class (philosophy)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005579396,0.002101327,0.0007409084,0.003401589,0.001263082,0.004154385,0.00461849,0.002022335,0.003624106],"category_scores_gemma":[0.03371415,0.001257693,0.00379475,0.001571691,0.003217078,0.006315918,0.005987109,0.00483763,0.0009848343],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002347311,"about_ca_system_score_gemma":0.004681072,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02779943,"about_ca_topic_score_gemma":0.04556472,"domain_scores_codex":[0.995486,0.001768382,0.0003902814,0.001198741,0.0008670071,0.0002895632],"domain_scores_gemma":[0.981568,0.01123591,0.00180302,0.003616009,0.001265247,0.0005117733],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008028411,0.0005275309,0.0204537,0.001845287,0.0007836515,0.001132972,0.004769893,0.3083536,0.01173585,0.3732047,0.0254739,0.2509161],"study_design_scores_gemma":[0.00007627688,0.00006288238,0.001293838,0.0001242536,0.00009012898,0.000169668,0.0002091186,0.6719674,0.003620601,0.3044553,0.01787056,0.00005993507],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.005918931,0.0003610344,0.9802489,0.0005831335,0.00004351262,0.0002036087,0.001440479,0.009872527,0.001327868],"genre_scores_gemma":[0.1617336,0.000400537,0.8310232,0.0003780103,0.00008195651,0.0004499281,0.003851176,0.0008566061,0.001225039],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02779943,"threshold_uncertainty_score":0.05527526,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07947094746768106,"score_gpt":0.3715199134609973,"score_spread":0.2920489659933162,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}