{"id":"W3030521948","doi":"","title":"HardEval: Focusing on Challenging Tokens to Assess Robustness of NER.","year":2020,"lang":"en","type":"article","venue":"NPARC","topic":"Topic Modeling","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"Institut de Valorisation des Données","keywords":"Robustness (evolution); Computer science; Ambiguity; Exploit; Artificial intelligence; Machine learning; Natural language processing; Programming language; Computer security","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01667534,0.003055381,0.001699737,0.007067733,0.002093524,0.003762142,0.003186245,0.003117535,0.004676537],"category_scores_gemma":[0.07067739,0.0006165184,0.001631938,0.003814126,0.002384532,0.00847316,0.005765129,0.003673223,0.003518142],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001454452,"about_ca_system_score_gemma":0.001366174,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004560472,"about_ca_topic_score_gemma":0.00788839,"domain_scores_codex":[0.9827372,0.00825299,0.001463578,0.003379206,0.003621273,0.0005457695],"domain_scores_gemma":[0.9272116,0.04873139,0.003502834,0.01204382,0.007209625,0.001300754],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003142124,0.001422968,0.08290946,0.005328585,0.00280934,0.0009735164,0.002400632,0.230269,0.03997082,0.01654422,0.07363366,0.5405958],"study_design_scores_gemma":[0.0002552371,0.002011325,0.05001311,0.0004155621,0.0007012152,0.002053556,0.001966116,0.7564154,0.1019239,0.04092225,0.04283432,0.0004879187],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2792453,0.006067986,0.63672,0.001719133,0.00156105,0.001842288,0.01462507,0.02587393,0.03234519],"genre_scores_gemma":[0.7285354,0.0007565034,0.2213133,0.0008828898,0.000334872,0.001360068,0.0343361,0.004525541,0.007955425],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01667534,"threshold_uncertainty_score":0.08818865,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1133778731560633,"score_gpt":0.2902726815032783,"score_spread":0.176894808347215,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}