{"id":"W2579310195","doi":"","title":"Artificial Intelligence Testing","year":2016,"lang":"en","type":"article","venue":"The Florida AI Research Society","topic":"Ethics and Social Impacts of AI","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Turing test; Test (biology); Deception; Argument (complex analysis); Subjectivity; Computer science; Artificial intelligence; Turing; Key (lock); Outcome (game theory); Self-deception; Cognitive science; Epistemology; Psychology; Social psychology; Computer security; Mathematics; Philosophy; Mathematical economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","sts"],"consensus_categories":["sts"],"category_scores_codex":[0.01627987,0.0000975876,0.0001211347,0.00002930732,0.004165659,0.0004396151,0.0009277092,0.0001934957,0.0002668144],"category_scores_gemma":[0.0130253,0.00005428226,0.0001529961,0.001028882,0.003082574,0.0004518415,0.0002449915,0.000861969,0.0003928113],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000298189,"about_ca_system_score_gemma":0.0009689356,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003779485,"about_ca_topic_score_gemma":0.0008534772,"domain_scores_codex":[0.9960654,0.00084192,0.0002162343,0.0002546885,0.00161868,0.001003012],"domain_scores_gemma":[0.9923345,0.005635475,0.00004671821,0.000335588,0.001395487,0.0002522273],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00001628187,0.00004851584,0.000616752,0.00001014041,0.00004211012,0.000004395258,0.07009836,0.000003435077,0.004239831,0.7102821,0.04035798,0.17428],"study_design_scores_gemma":[0.00004292878,0.00009219638,0.0004778624,0.00007658611,0.00000654124,5.142832e-7,0.02705947,0.000119399,0.002296775,0.8965963,0.07302961,0.0002018425],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.163029,0.0002444341,0.004671449,0.7051252,0.00141971,0.001006648,0.00002198863,0.0003630957,0.1241185],"genre_scores_gemma":[0.9914933,0.0004972516,0.0004011493,0.001101051,0.00286942,0.00002527661,3.505582e-7,0.00001679864,0.003595412],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8284643,"threshold_uncertainty_score":0.9996305,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4743838374723485,"score_gpt":0.5305625539451253,"score_spread":0.05617871647277684,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}