{"id":"W4308835006","doi":"10.1002/pds.5555","title":"More extreme duplication in FDA Adverse Event Reporting System detected by literature reference normalization and fuzzy string matching","year":2022,"lang":"en","type":"article","venue":"Pharmacoepidemiology and Drug Safety","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":22,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Adverse Event Reporting System; Computer science; Data mining; Information retrieval; Levenshtein distance; Normalization (sociology); Medicine; Matching (statistics); String metric; String searching algorithm; Adverse effect; Artificial intelligence; Pattern matching; Internal medicine","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002262473,0.000146766,0.0002412146,0.00005653515,0.0002666216,0.00000598928,0.0001001228,0.0001291324,0.000007135906],"category_scores_gemma":[0.0005666526,0.0001373105,0.00003279015,0.0001339916,0.00007766193,0.000008420116,0.000231452,0.0003337149,3.851641e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003935246,"about_ca_system_score_gemma":0.00002663725,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00009978029,"about_ca_topic_score_gemma":0.00001677767,"domain_scores_codex":[0.9979311,0.0006053726,0.0007003397,0.0004461383,0.00007708953,0.000240021],"domain_scores_gemma":[0.999128,0.0001215518,0.0004959685,0.0001460886,0.00003133136,0.00007708601],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","study_design_scores_codex":[0.001481982,0.0001706788,0.3397068,0.0007649208,0.0001743114,0.0000989543,0.003320414,0.0038484,0.5335839,0.001299647,0.004030903,0.1115191],"study_design_scores_gemma":[0.01300918,0.0009318879,0.6749087,0.001165448,0.0003904407,0.002866246,0.02326753,0.05009244,0.04083994,0.001948052,0.1869824,0.003597756],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9882368,0.006332746,0.00411706,0.0008180062,0.0001255796,0.0001755912,0.00004223562,0.00004499262,0.0001070244],"genre_scores_gemma":[0.9973357,0.0009241675,0.0006795403,0.0003669412,0.00005319967,0.0000603873,0.0004859141,0.000009737632,0.00008446369],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.492744,"threshold_uncertainty_score":0.5599362,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02566215284907491,"score_gpt":0.3131290756210826,"score_spread":0.2874669227720076,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}