{"id":"W4409363778","doi":"10.1609/aaai.v39i24.34710","title":"Tuning-Free Accountable Intervention for LLM Deployment – a Metacognitive Approach","year":2025,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multi-Agent Systems and Negotiation","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Software deployment; Metacognition; Intervention (counseling); Psychology; Psychotherapist; Computer science; Neuroscience; Cognition; Psychiatry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008466275,0.0002039938,0.0002753267,0.0001988929,0.0002352362,0.0003203299,0.00191316,0.00008685303,0.00001621267],"category_scores_gemma":[0.000507595,0.0001560483,0.0002070387,0.0006031776,0.00008968425,0.0005452915,0.0004711766,0.0001584993,0.00001487182],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007973677,"about_ca_system_score_gemma":0.00007411694,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001162936,"about_ca_topic_score_gemma":0.00001826092,"domain_scores_codex":[0.9982055,0.00002405858,0.0005978672,0.0005198388,0.0003609826,0.0002917684],"domain_scores_gemma":[0.9981962,0.0001049564,0.0004400204,0.0003461511,0.0008662287,0.00004642373],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00007674335,0.0004009705,0.0001134522,0.0001839411,0.00005779047,5.364609e-8,0.001054783,0.00003728866,0.01527258,0.9382647,0.0007330389,0.04380461],"study_design_scores_gemma":[0.0001731427,0.0002421449,0.0002331397,0.0006147953,0.00005416795,0.000001419903,0.001226742,0.314111,0.5021158,0.180534,0.0004518003,0.000241819],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02367344,0.00005990743,0.9578586,0.001854179,0.0008004681,0.00149122,0.00001265727,0.0001124163,0.01413708],"genre_scores_gemma":[0.9892161,0.00001324384,0.009060289,0.0001835007,0.00004622518,0.0002514886,0.000001732764,0.000009439591,0.001217954],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9655427,"threshold_uncertainty_score":0.6363465,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0874622275211003,"score_gpt":0.3196241001394326,"score_spread":0.2321618726183323,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}