{"id":"W7125908856","doi":"10.1109/ase63991.2025.00067","title":"Watson: A Cognitive Observability Framework for the Reasoning of LLM-Powered Agents","year":2025,"lang":"","type":"article","venue":"","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University; Huawei Technologies (Canada)","funders":"","keywords":"Observability; Debugging; Transparency (behavior); Software; Benchmark (surveying); Semantic reasoner; Autonomous agent; Cognition; Model-based reasoning; Class (philosophy)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.002554477,0.0004628286,0.0006670976,0.0001413602,0.0009630133,0.0004737061,0.002619937,0.0003630746,0.0005176837],"category_scores_gemma":[0.009150789,0.000356549,0.0005182738,0.002240825,0.0007566226,0.0006887201,0.00119834,0.0005668688,0.00007194756],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001711216,"about_ca_system_score_gemma":0.0007000282,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008056013,"about_ca_topic_score_gemma":0.0001903313,"domain_scores_codex":[0.9954433,0.0003840267,0.001266725,0.001265812,0.0005957458,0.001044374],"domain_scores_gemma":[0.9864041,0.009425479,0.0004789244,0.001835137,0.001684745,0.0001716104],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003575222,0.0007710339,0.003313587,0.0002803131,0.0003453745,0.000006197864,0.006839274,0.0003067206,0.0001132559,0.8360815,0.000995971,0.1505892],"study_design_scores_gemma":[0.0004502523,0.0006880009,0.007628677,0.001925247,0.0002803144,0.000002363009,0.0109757,0.5351083,0.08052657,0.3580285,0.003799635,0.000586473],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02353514,0.001622155,0.9579315,0.005898332,0.002574629,0.002614748,0.00003278012,0.00009896921,0.005691812],"genre_scores_gemma":[0.9340875,0.0001985583,0.06086724,0.001965455,0.0001126914,0.0002297953,0.000002698396,0.00001982406,0.002516198],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9105524,"threshold_uncertainty_score":0.9998887,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07947094746768106,"score_gpt":0.3715199134609973,"score_spread":0.2920489659933162,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}