{"id":"W7138047150","doi":"10.18653/v1/2025.banglalp-1.14","title":"BLUCK: A Benchmark Dataset for Bengali Linguistic Understanding and Cultural Knowledge","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Bengali; Benchmark (surveying); Cultural knowledge; Knowledge-based systems; Knowledge base; Computational linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00168036,0.002721439,0.0009717172,0.005944312,0.002110473,0.002208316,0.003821703,0.002266687,0.01566863],"category_scores_gemma":[0.00923968,0.0003693614,0.001384845,0.006107106,0.0009651051,0.003651971,0.003330994,0.00191209,0.01394854],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004258421,"about_ca_system_score_gemma":0.003067414,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.1310235,"about_ca_topic_score_gemma":0.1548204,"domain_scores_codex":[0.9975882,0.000772534,0.0002974813,0.0005362183,0.0004963989,0.00030931],"domain_scores_gemma":[0.9965912,0.0009265666,0.0001855244,0.0008813688,0.001122947,0.0002923707],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0009274405,0.0006925696,0.02164942,0.002554679,0.0003521444,0.0007619534,0.002803257,0.008363171,0.006722561,0.005860431,0.7591277,0.1901847],"study_design_scores_gemma":[0.0002585434,0.0003016977,0.06010212,0.0006757769,0.0003044185,0.001120018,0.007488113,0.05644781,0.01539377,0.008167408,0.8494508,0.0002895262],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.1162902,0.005200211,0.01801922,0.002838109,0.0007098853,0.001516116,0.7939955,0.02242588,0.03900479],"genre_scores_gemma":[0.06501552,0.0005242593,0.01332201,0.0004797642,0.00007805286,0.0007736402,0.9123678,0.0004935502,0.006945377],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.1310235,"threshold_uncertainty_score":0.2605217,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05115238458861413,"score_gpt":0.3614718791674825,"score_spread":0.3103194945788684,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}