{"id":"W4416713956","doi":"10.36227/techrxiv.175735299.97215847/v3","title":"Responsible Agentic Reasoning and AI Agents: A Critical Survey","year":2025,"lang":"","type":"article","venue":"","topic":"Ethics and Social Impacts of AI","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute","funders":"HORIZON EUROPE Framework Programme; European Commission","keywords":"Operationalization; Intersection (aeronautics); Audit; Trustworthiness; Automated reasoning; Soundness; Class (philosophy)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01834156,0.0009548191,0.001324226,0.009163043,0.001620183,0.008132632,0.00217112,0.003849862,0.002769955],"category_scores_gemma":[0.03329536,0.001129191,0.0008733959,0.007762376,0.006844997,0.01647116,0.003229456,0.004823295,0.0009955956],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005693525,"about_ca_system_score_gemma":0.005343616,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00404714,"about_ca_topic_score_gemma":0.002276364,"domain_scores_codex":[0.9839892,0.007667247,0.001520718,0.001212972,0.005015679,0.0005942878],"domain_scores_gemma":[0.9418239,0.04466316,0.001830074,0.00277263,0.008038782,0.0008713508],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0000601337,0.0001124395,0.003080828,0.003870959,0.0000766365,0.00009884887,0.001479645,0.004612698,0.0002409014,0.5949529,0.01368751,0.3777265],"study_design_scores_gemma":[0.00001732527,0.0001366796,0.002101871,0.006508369,0.00007834672,0.0004465486,0.002385373,0.01438613,0.001005799,0.526384,0.4464678,0.00008183109],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.005578819,0.794196,0.1267283,0.03830941,0.0008858853,0.0001617984,0.0001526514,0.0003611847,0.03362585],"genre_scores_gemma":[0.126868,0.7689093,0.092439,0.00463226,0.002771128,0.0003662705,0.0003059091,0.0002444516,0.003463746],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.01834156,"threshold_uncertainty_score":0.0970006,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07530069601208889,"score_gpt":0.4645475968993344,"score_spread":0.3892469008872456,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}