{"id":"W4385979136","doi":"10.1145/3604570","title":"The End of the Policy Analyst? Testing the Capability of Artificial Intelligence to Generate Plausible, Persuasive, and Useful Policy Analysis","year":2023,"lang":"en","type":"article","venue":"Digital Government Research and Practice","topic":"E-Government and Public Services","field":"Social Sciences","cited_by":29,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Regina; Shared Services Canada","funders":"","keywords":"Rubric; Computer science; Government (linguistics); Artificial intelligence; Task (project management); Heuristic; Public policy; Quality (philosophy); Political science; Psychology; Engineering; Mathematics education; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.006406941,0.000104264,0.0001595387,0.0001085152,0.001216362,0.0007637153,0.0006250219,0.00004770289,0.00001578638],"category_scores_gemma":[0.02762462,0.00005681792,0.00007179781,0.005991144,0.001437293,0.0006226487,0.000653395,0.0002220013,0.000004974078],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002084774,"about_ca_system_score_gemma":0.0004982474,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02951271,"about_ca_topic_score_gemma":0.01146751,"domain_scores_codex":[0.9957656,0.0007384706,0.0003154249,0.0002758277,0.002435029,0.0004696686],"domain_scores_gemma":[0.9872559,0.01160938,0.0002393721,0.0003478476,0.0003819345,0.0001656051],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","study_design_scores_codex":[0.0008009841,0.0003495159,0.2523288,0.00008435307,0.001184617,0.000004890991,0.03592314,0.0004127739,0.001421089,0.5384807,0.0008207951,0.1681883],"study_design_scores_gemma":[0.0001794368,0.0007592816,0.2575374,0.00006413465,0.0003501668,0.000005144293,0.5848106,0.003222084,0.004232984,0.03992157,0.1084522,0.000465043],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8764838,0.0001163781,0.00002317208,0.07075981,0.00003064573,0.0004170907,0.0001444242,0.00001142775,0.05201326],"genre_scores_gemma":[0.9982237,0.0003865699,0.00004637853,0.0001664376,0.0001640783,0.00001763878,0.00000172535,0.000005683408,0.0009877448],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5488875,"threshold_uncertainty_score":0.9805661,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1653987492323391,"score_gpt":0.4347940881314314,"score_spread":0.2693953388990923,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}