{"id":"W4405034837","doi":"10.48550/arxiv.2412.01621","title":"NYT-Connections: A Deceptively Simple Text Classification Task that Stumps System-1 Thinkers","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs","keywords":"Simple (philosophy); Task (project management); Computer science; Linguistics; Natural language processing; Psychology; Cognitive psychology; Epistemology; Philosophy; Economics; Management","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0004039137,0.0005179163,0.0005326914,0.0006829394,0.0003105212,0.0003728688,0.002432518,0.000440511,0.00002085912],"category_scores_gemma":[0.00004593489,0.0005221053,0.0004779274,0.00159501,0.0001306374,0.0006425345,0.003012463,0.001012582,0.0003554773],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0012845,"about_ca_system_score_gemma":0.0002404987,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003000676,"about_ca_topic_score_gemma":0.0002125135,"domain_scores_codex":[0.9967175,0.0002478403,0.0003509422,0.002017813,0.0002205154,0.0004453603],"domain_scores_gemma":[0.9970357,0.0002208995,0.0004895394,0.001783585,0.0002780498,0.0001922716],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002165188,0.0001247372,0.001112233,0.0003087823,0.0005073769,0.0003225218,0.0005631094,0.02279305,0.0005808657,0.9667823,0.002754268,0.004129154],"study_design_scores_gemma":[0.0001998962,0.00005947121,0.0008372338,0.0003431882,0.0003526424,0.00001586126,0.001244882,0.7261115,0.0005226756,0.2669666,0.002481247,0.000864837],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05431323,0.0002365442,0.937231,0.0002267023,0.0005103783,0.0006558057,0.00004427236,0.002999368,0.003782719],"genre_scores_gemma":[0.9918929,0.0001906579,0.006466003,0.00006962541,0.00009584194,0.00001516742,0.00005477258,0.00003129224,0.001183773],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9375796,"threshold_uncertainty_score":0.9997231,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08503281301773807,"score_gpt":0.223244329763299,"score_spread":0.1382115167455609,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}