{"id":"W4405034837","doi":"10.48550/arxiv.2412.01621","title":"NYT-Connections: A Deceptively Simple Text Classification Task that Stumps System-1 Thinkers","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs","keywords":"Simple (philosophy); Task (project management); Computer science; Linguistics; Natural language processing; Psychology; Cognitive psychology; Epistemology; Philosophy; Economics; Management","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002519974,0.002039622,0.0007465656,0.001149935,0.001008343,0.00191753,0.00241897,0.002326587,0.009192308],"category_scores_gemma":[0.0349698,0.0004720587,0.0007433766,0.0008582106,0.001200541,0.005011485,0.002709112,0.002568329,0.004158956],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001198671,"about_ca_system_score_gemma":0.001848211,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007192834,"about_ca_topic_score_gemma":0.0136612,"domain_scores_codex":[0.9975699,0.0008405023,0.0002376615,0.0006396571,0.0005587345,0.0001535342],"domain_scores_gemma":[0.9802256,0.01482485,0.0009307107,0.002395533,0.0009692467,0.0006539983],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005203388,0.001960159,0.03076324,0.003334309,0.0006184004,0.001229055,0.002808936,0.07930251,0.0244183,0.02543011,0.2579723,0.5669593],"study_design_scores_gemma":[0.0007905992,0.001619272,0.009817233,0.0002171381,0.0001240014,0.0007900727,0.0009632259,0.8367217,0.02684575,0.04706456,0.07485582,0.0001906371],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7470392,0.001678147,0.1220854,0.003602339,0.001505153,0.001512849,0.01915306,0.05972926,0.04369448],"genre_scores_gemma":[0.8102795,0.0003232314,0.1476618,0.001028364,0.000196843,0.0008334828,0.02255523,0.002556911,0.01456447],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009192308,"threshold_uncertainty_score":0.03075135,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08503281301773807,"score_gpt":0.223244329763299,"score_spread":0.1382115167455609,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}