{"id":"W4366201187","doi":"10.1109/access.2023.3267746","title":"B-NER: A Novel Bangla Named Entity Recognition Dataset With Largest Entities and Its Baseline Evaluation","year":2023,"lang":"en","type":"article","venue":"IEEE Access","topic":"Topic Modeling","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Named-entity recognition; Computer science; Bengali; Natural language processing; Artificial intelligence; Entity linking; Baseline (sea); Benchmark (surveying); Sentence; F1 score; Task (project management); Information retrieval","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002922909,0.001945444,0.001412607,0.003642695,0.002263663,0.00191853,0.003320856,0.002521508,0.005100152],"category_scores_gemma":[0.006926817,0.0003807592,0.001686015,0.003413919,0.001004374,0.003824531,0.002486517,0.001634783,0.007357467],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0015975,"about_ca_system_score_gemma":0.001599946,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01363949,"about_ca_topic_score_gemma":0.02279222,"domain_scores_codex":[0.9963408,0.0006273408,0.0006908786,0.001285871,0.0008168415,0.000238231],"domain_scores_gemma":[0.9959136,0.000832755,0.000298431,0.001545834,0.001183065,0.0002262493],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002388242,0.002601809,0.02311997,0.007465513,0.0007224029,0.002293665,0.001081771,0.02523502,0.04516461,0.008774351,0.5869299,0.2942227],"study_design_scores_gemma":[0.0005572434,0.0010627,0.07530535,0.001014979,0.0005292509,0.005627116,0.002630175,0.1188923,0.06275047,0.007030364,0.7240108,0.0005893386],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.227488,0.00600824,0.06450721,0.002110098,0.002048588,0.002942279,0.6239859,0.03343116,0.0374785],"genre_scores_gemma":[0.06205065,0.0006745115,0.05637589,0.0005076252,0.0001027079,0.001330806,0.8719398,0.0005598449,0.006458157],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01363949,"threshold_uncertainty_score":0.02712017,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1369707390250249,"score_gpt":0.3440463260481199,"score_spread":0.207075587023095,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}