{"id":"W4401588373","doi":"10.1016/j.paid.2024.112840","title":"Hacking the perfect score on high-stakes personality assessments with generative AI","year":2024,"lang":"en","type":"article","venue":"Personality and Individual Differences","topic":"Personality Traits and Psychology","field":"Psychology","cited_by":14,"is_retracted":false,"has_abstract":false,"ca_institutions":"Wilfrid Laurier University","funders":"Social Sciences and Humanities Research Council of Canada; Canadian Psychological Association","keywords":"Psychology; Hacker; Generative grammar; Personality; Personality test; Social psychology; Cognitive psychology; Clinical psychology; Psychometrics; Artificial intelligence; Test validity; Computer security; Computer science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005285283,0.0005697547,0.0004299732,0.001458265,0.0004200225,0.001569005,0.000472607,0.0007408437,0.003600514],"category_scores_gemma":[0.04771461,0.0002451641,0.000477112,0.000593347,0.0007078879,0.001501202,0.001869218,0.001039787,0.001570495],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003380036,"about_ca_system_score_gemma":0.0003837605,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001150051,"about_ca_topic_score_gemma":0.002651056,"domain_scores_codex":[0.9963218,0.001452586,0.0003568152,0.000515198,0.001030515,0.0003231223],"domain_scores_gemma":[0.9786912,0.0110251,0.002279627,0.00426012,0.002651203,0.001092746],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001719436,0.0007797304,0.6414148,0.00009437811,0.000231593,0.0002435553,0.002686132,0.003628359,0.006780526,0.004398175,0.007942074,0.3300811],"study_design_scores_gemma":[0.00009571338,0.001904916,0.9071544,0.0001313635,0.0001349655,0.0009614405,0.001944986,0.04751606,0.0153006,0.01797135,0.006733649,0.0001506255],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9673639,0.00008199393,0.014076,0.0002515581,0.0002147365,0.0001556161,0.0003796225,0.0008566907,0.01661985],"genre_scores_gemma":[0.9905744,0.00003604952,0.006974498,0.00006404857,0.00001834782,0.0000771226,0.0002014579,0.00005342545,0.002000776],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005285283,"threshold_uncertainty_score":0.02795154,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1420897258098148,"score_gpt":0.3780939606685988,"score_spread":0.2360042348587841,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}