{"id":"W3037980084","doi":"10.1109/cist49399.2021.9357170","title":"Leveraging Subword Embeddings for Multinational Address Parsing","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université Laval","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Parsing; Computer science; Python (programming language); Artificial intelligence; Disk formatting; Natural language processing; Artificial neural network; Machine learning; Programming language","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002925494,0.0002194218,0.0002568695,0.00009861797,0.0001302956,0.0004152914,0.001310781,0.0001433137,0.000016188],"category_scores_gemma":[0.0001366859,0.0002318862,0.0001553095,0.0000899293,0.00001490286,0.0002130414,0.001894925,0.0003663318,0.00001476508],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001007041,"about_ca_system_score_gemma":0.0002060612,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004192088,"about_ca_topic_score_gemma":0.000002907911,"domain_scores_codex":[0.9981008,0.00002919551,0.0003376066,0.0009007582,0.0003338671,0.0002977468],"domain_scores_gemma":[0.9988545,0.0001862286,0.0001580142,0.0005226212,0.0001744974,0.0001040939],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002823558,0.0001284546,0.0008863221,0.001240606,0.0002148056,0.00003463611,0.01278308,0.1373973,0.001247018,0.5150431,0.008880001,0.3221163],"study_design_scores_gemma":[0.0002156947,0.000006672619,0.0001362308,0.00009778885,0.000006797054,0.000002809814,0.00002482427,0.9468279,0.000569282,0.0486512,0.003190626,0.0002702014],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003146509,0.00005824875,0.9853262,0.007892468,0.001196471,0.0004329586,0.000006279952,0.0004127461,0.00152809],"genre_scores_gemma":[0.4123562,0.000002189627,0.5862319,0.0007660591,0.0002972327,0.00005975353,0.00002433112,0.00001382981,0.0002485311],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8094305,"threshold_uncertainty_score":0.9456046,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09820144419307504,"score_gpt":0.3210443635176114,"score_spread":0.2228429193245364,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}