{"id":"W4392425838","doi":"10.48550/arxiv.2403.00126","title":"FAC$^2$E: Better Understanding Large Language Model Capabilities by Dissociating Language and Cognition","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Cognition; Psychology; Cognitive psychology; Computer science; Cognitive science; Linguistics; Natural language processing; Philosophy; Neuroscience","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004297504,0.00149197,0.0006288826,0.002221229,0.0005538392,0.00388554,0.00128295,0.001281257,0.007728222],"category_scores_gemma":[0.02302855,0.0004256248,0.00103373,0.001110466,0.0009011329,0.008042537,0.003917548,0.001860092,0.001734751],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001084834,"about_ca_system_score_gemma":0.001929698,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006782713,"about_ca_topic_score_gemma":0.009608395,"domain_scores_codex":[0.9981776,0.0008206005,0.0001430365,0.0003298347,0.000405363,0.0001235489],"domain_scores_gemma":[0.9870595,0.007773398,0.0007029102,0.003184318,0.0009514911,0.0003283618],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007010744,0.0005778712,0.03163393,0.0006418541,0.0003850469,0.0001711801,0.001584035,0.09031087,0.01749865,0.06452893,0.01392075,0.7780458],"study_design_scores_gemma":[0.00005556108,0.0002894208,0.0105685,0.00007985931,0.0001195939,0.0002218346,0.0005628031,0.8443502,0.02427715,0.1085114,0.01083299,0.0001308535],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0962551,0.0005145722,0.8782799,0.001110834,0.00007006298,0.0002995246,0.001391045,0.01061835,0.01146054],"genre_scores_gemma":[0.5973754,0.0002377617,0.3961288,0.0002777366,0.00004247444,0.0002499227,0.002333247,0.0006115693,0.002743111],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007728222,"threshold_uncertainty_score":0.02585346,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0616744407948395,"score_gpt":0.2025338497986467,"score_spread":0.1408594090038072,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}