{"id":"W49270455","doi":"","title":"Using Unigram and Bigram Language Models for Monolingual and Cross-Language IR","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Bigram; Computer science; Natural language processing; Artificial intelligence; Word (group theory); Search engine indexing; Machine translation; Cross-language information retrieval; Translation (biology); Character (mathematics); Speech recognition; Linguistics; Mathematics; Trigram","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006516594,0.001462226,0.001319005,0.002721249,0.0009276019,0.002150155,0.0009017732,0.001132434,0.003282035],"category_scores_gemma":[0.01097731,0.0004100497,0.0009742968,0.001518904,0.0004374643,0.006642491,0.001315928,0.001279633,0.002772964],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008533457,"about_ca_system_score_gemma":0.0009073828,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005201311,"about_ca_topic_score_gemma":0.006810311,"domain_scores_codex":[0.9969522,0.002000577,0.0001732192,0.0003667441,0.0003672715,0.0001400266],"domain_scores_gemma":[0.9933167,0.005013388,0.000222161,0.0006015195,0.0007054845,0.0001407475],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002221979,0.0007369847,0.005622986,0.0008493849,0.0007657225,0.0002940917,0.0006874813,0.07519393,0.01480817,0.00759839,0.00695177,0.8842692],"study_design_scores_gemma":[0.0001206908,0.0006213255,0.002328268,0.00005304612,0.0002437147,0.0003061605,0.000329671,0.9667317,0.01135647,0.01372425,0.004052565,0.0001322328],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2711961,0.008025695,0.6935802,0.0006032787,0.0003964211,0.0003622102,0.0007058073,0.009965812,0.0151646],"genre_scores_gemma":[0.7174289,0.001915965,0.268512,0.0002960265,0.0002235116,0.0002892187,0.001308342,0.0006440549,0.009382006],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006516594,"threshold_uncertainty_score":0.03446347,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03112713027741485,"score_gpt":0.3634326685443653,"score_spread":0.3323055382669505,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}