{"id":"W4399117309","doi":"10.3758/s13428-024-02441-0","title":"Shadows of wisdom: Classifying meta-cognitive and morally grounded narrative content via large language models","year":2024,"lang":"en","type":"article","venue":"Behavior Research Methods","topic":"Psychology of Moral and Emotional Judgment","field":"Neuroscience","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Social Sciences and Humanities Research Council of Canada; John Templeton Foundation","keywords":"Narrative; Categorization; Humility; Classifier (UML); Psychology; Workflow; Coding (social sciences); Computer science; Social psychology; Artificial intelligence; Sociology; Linguistics; Social science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006905064,0.0009864329,0.0003900476,0.00211016,0.0007842063,0.002711133,0.001464699,0.0008816893,0.001705672],"category_scores_gemma":[0.03598443,0.0003436034,0.0009919986,0.0007232739,0.0009404151,0.003027476,0.002260106,0.001759579,0.0007861715],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002172181,"about_ca_system_score_gemma":0.002008436,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01270488,"about_ca_topic_score_gemma":0.02338277,"domain_scores_codex":[0.9971027,0.001831771,0.0001514661,0.0004819013,0.0003266942,0.0001056228],"domain_scores_gemma":[0.9745763,0.02091941,0.001055085,0.001521741,0.001506076,0.0004213842],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001711485,0.0007550239,0.1177857,0.001555465,0.0005444236,0.0009519117,0.03815889,0.07284591,0.02653965,0.01937557,0.01423641,0.7055396],"study_design_scores_gemma":[0.00007330141,0.00021015,0.0252218,0.0002945621,0.0001502419,0.0002702461,0.00634011,0.9062245,0.01415256,0.03514812,0.01176534,0.000149009],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.64324,0.0008679848,0.3394935,0.001321358,0.0001181388,0.0007198249,0.00391915,0.004636403,0.005683539],"genre_scores_gemma":[0.8272589,0.000129713,0.1677324,0.0001518616,0.00002164138,0.0003553272,0.003141368,0.000189211,0.001019499],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01270488,"threshold_uncertainty_score":0.03651792,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7650757583834222,"score_gpt":0.5756139618117823,"score_spread":0.18946179657164,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}