{"id":"W4401588373","doi":"10.1016/j.paid.2024.112840","title":"Hacking the perfect score on high-stakes personality assessments with generative AI","year":2024,"lang":"en","type":"article","venue":"Personality and Individual Differences","topic":"Personality Traits and Psychology","field":"Psychology","cited_by":14,"is_retracted":false,"has_abstract":false,"ca_institutions":"Wilfrid Laurier University","funders":"Social Sciences and Humanities Research Council of Canada; Canadian Psychological Association","keywords":"Psychology; Hacker; Generative grammar; Personality; Personality test; Social psychology; Cognitive psychology; Clinical psychology; Psychometrics; Artificial intelligence; Test validity; Computer security; Computer science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001012188,0.0004547373,0.0004380958,0.0001117913,0.0007025622,0.0006555027,0.0004203984,0.0002159568,0.003277583],"category_scores_gemma":[0.00002340847,0.0002550759,0.000139241,0.0003046716,0.0009593182,0.0002330558,0.00005948216,0.0009124426,0.00007807872],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000416588,"about_ca_system_score_gemma":0.0001046247,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001423717,"about_ca_topic_score_gemma":0.000537527,"domain_scores_codex":[0.996573,0.0008004034,0.0003020452,0.001028121,0.0006981264,0.0005982618],"domain_scores_gemma":[0.9986152,0.0007171454,0.00008989697,0.0003428433,0.00006732489,0.0001676133],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0007531277,0.0008218116,0.4425402,0.0002618862,0.002868034,0.0001384548,0.1007662,0.000003271693,0.00009393856,0.377586,0.01432462,0.05984253],"study_design_scores_gemma":[0.0005806248,0.0009612865,0.9831803,0.0001748459,0.0001976715,0.00008531545,0.005053431,0.00007860015,0.00001634254,0.003534659,0.005721861,0.0004150545],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9704672,0.001953489,0.0002901927,0.01235659,0.0008321194,0.0003321182,0.0007743022,0.0001294519,0.01286453],"genre_scores_gemma":[0.9910644,0.00004937215,0.000095148,0.006623674,0.0005971421,0.0001203271,0.0001071832,0.00002830667,0.001314411],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5406401,"threshold_uncertainty_score":0.9999902,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1420897258098148,"score_gpt":0.3780939606685988,"score_spread":0.2360042348587841,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}