{"id":"W4404783146","doi":"10.18653/v1/2024.emnlp-main.953","title":"Measuring Psychological Depth in Language Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Science Foundation","keywords":"Computer science; Natural language processing; Language model; Artificial intelligence; Programming language","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003121931,0.0004535654,0.0003274475,0.001150516,0.0004504326,0.002159898,0.0004560355,0.0006694819,0.002245241],"category_scores_gemma":[0.05264314,0.000515043,0.0004463511,0.0009755628,0.001304623,0.005408834,0.002439097,0.001720389,0.0002768686],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007484523,"about_ca_system_score_gemma":0.0003672883,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003436364,"about_ca_topic_score_gemma":0.003022759,"domain_scores_codex":[0.9979538,0.001221729,0.00005904599,0.0002976911,0.0003622884,0.0001054968],"domain_scores_gemma":[0.9785729,0.01548907,0.001567721,0.002345155,0.001273517,0.0007516503],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001544842,0.0008085863,0.4124382,0.0004777907,0.0009375279,0.0002864701,0.02318572,0.06166996,0.01590572,0.163959,0.007618349,0.3111679],"study_design_scores_gemma":[0.00009420211,0.0004520371,0.245341,0.0001744131,0.000322828,0.0004683776,0.008523208,0.3114153,0.006394714,0.4195347,0.007051178,0.0002281156],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8994926,0.001137132,0.08454298,0.001173478,0.00003474704,0.00007186385,0.0004939632,0.0001789132,0.01287444],"genre_scores_gemma":[0.9929433,0.0001302159,0.006384766,0.00004723144,0.00000984435,0.00002406675,0.000136518,0.00002460389,0.0002994442],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003436364,"threshold_uncertainty_score":0.01651055,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2191795032686007,"score_gpt":0.4647030931911161,"score_spread":0.2455235899225154,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}