{"id":"W3153895709","doi":"10.18653/v1/2021.eacl-srw.21","title":"TMR: Evaluating NER Recall on Tough Mentions","year":2021,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Atomic Energy of Canada Limited","keywords":"Computer science; Recall; Natural language processing; Linguistics; Philosophy","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002158419,0.00005010001,0.00005586303,0.00002211924,0.00008009115,0.00009625876,0.0002400793,0.00002276223,0.0002882422],"category_scores_gemma":[0.00007704632,0.00004437637,0.00003761435,0.0001514429,0.000004265062,0.0001443617,0.0001719205,0.00006827561,0.0001578901],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002552874,"about_ca_system_score_gemma":0.00006088845,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001395124,"about_ca_topic_score_gemma":0.00001415149,"domain_scores_codex":[0.9992023,0.0000540426,0.0001240636,0.0002647772,0.0002199841,0.0001348491],"domain_scores_gemma":[0.9993419,0.00005939558,0.00001986025,0.0004705877,0.00006990232,0.00003835392],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[8.183341e-7,0.00007448137,0.0001054825,0.000005543367,0.00001454106,0.0000227413,0.0005636303,0.005144224,0.004481256,0.6805977,0.003816211,0.3051734],"study_design_scores_gemma":[0.0002507295,0.00004848336,0.0003633738,0.00002874182,0.000004918479,0.00001176595,0.00005035089,0.9609868,0.008010129,0.01938692,0.01070545,0.0001523386],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01940119,0.00003313035,0.8589996,0.004821595,0.0003889275,0.00004327691,1.688384e-7,0.0001263602,0.1161858],"genre_scores_gemma":[0.2892341,0.000004341705,0.6796852,0.002594552,0.0001166232,0.00001056373,0.000001518307,0.00000533504,0.02834771],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9558426,"threshold_uncertainty_score":0.315605,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1072639897763181,"score_gpt":0.3539379442665714,"score_spread":0.2466739544902533,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}