{"id":"W4379197269","doi":"10.1186/s12859-023-05350-9","title":"An analysis of entity normalization evaluation biases in specialized domains","year":2023,"lang":"en","type":"article","venue":"BMC Bioinformatics","topic":"Topic Modeling","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Normalization (sociology); Computer science; Data science; Task (project management); Natural language processing; Field (mathematics); Information retrieval; Artificial intelligence; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.08764417,0.001623071,0.001500503,0.004514329,0.00178181,0.003799562,0.001868487,0.001875577,0.001960711],"category_scores_gemma":[0.2436798,0.000500485,0.001091335,0.005996631,0.001864212,0.004881543,0.003380021,0.001930663,0.001039725],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002181054,"about_ca_system_score_gemma":0.001595846,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00271429,"about_ca_topic_score_gemma":0.003223612,"domain_scores_codex":[0.9198045,0.05100784,0.007143087,0.009075655,0.0115747,0.001394157],"domain_scores_gemma":[0.6308101,0.3110639,0.01006384,0.02025882,0.02647473,0.001328586],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004988856,0.0008775383,0.2380438,0.007117204,0.003276199,0.0009899318,0.005114113,0.03536004,0.0252003,0.01464476,0.08002727,0.5843599],"study_design_scores_gemma":[0.0007856508,0.002082117,0.3624359,0.003620936,0.003607666,0.005115466,0.004959687,0.3185002,0.1382,0.05870578,0.1014043,0.0005823125],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7629929,0.03073791,0.1664035,0.005215631,0.001110072,0.0009924578,0.009217371,0.006625447,0.0167048],"genre_scores_gemma":[0.9261284,0.001458636,0.0521949,0.001035149,0.0002990049,0.0007129041,0.01512445,0.001378566,0.00166801],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.08764417,"threshold_uncertainty_score":0.4635122,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08364742459170624,"score_gpt":0.3394736568482408,"score_spread":0.2558262322565346,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}