{"id":"W4416082371","doi":"10.36227/techrxiv.175735299.97215847/v2","title":"Responsible Agentic Reasoning and AI Agents: A Critical Survey","year":2025,"lang":"","type":"preprint","venue":"","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute","funders":"HORIZON EUROPE Framework Programme; European Commission","keywords":"Trustworthiness; Audit; Automated reasoning; Perception; Knowledge representation and reasoning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","scholarly_communication","open_science","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.007795226,0.001013133,0.001205693,0.0009232396,0.001111383,0.003729577,0.003237425,0.0008380412,0.001276011],"category_scores_gemma":[0.01545499,0.001117868,0.000314683,0.00210743,0.000896273,0.001340511,0.01016007,0.001977244,0.0006844953],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003548525,"about_ca_system_score_gemma":0.002812057,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008074521,"about_ca_topic_score_gemma":0.002490519,"domain_scores_codex":[0.9885787,0.0032309,0.001772788,0.003488841,0.001108003,0.001820739],"domain_scores_gemma":[0.9886796,0.005235469,0.0002696452,0.003238425,0.001762882,0.0008139104],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003741393,0.0007369519,0.008712904,0.001181929,0.0002985862,0.0006548481,0.005326456,0.001715239,0.0001955181,0.928783,0.01476014,0.03726027],"study_design_scores_gemma":[0.000254018,0.0003342407,0.02076815,0.002585014,0.0001555077,0.00007274831,0.000655991,0.869245,0.008953094,0.08781573,0.007175035,0.00198548],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01425504,0.001586582,0.9437934,0.009332994,0.004972751,0.001224657,0.0000785314,0.0003924624,0.02436364],"genre_scores_gemma":[0.8608889,0.001473355,0.08046466,0.006427888,0.0002580017,0.0001691219,0.00003556852,0.00006758054,0.05021494],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8675297,"threshold_uncertainty_score":0.9996369,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08589055885772025,"score_gpt":0.3903731380859501,"score_spread":0.3044825792282299,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}