{"id":"W1726342440","doi":"","title":"Identifying Parallel Documents from a Large Bilingual Collection of Texts: Application to Parallel Article Extraction in Wikipedia.","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Parallel corpora; Baseline (sea); Natural language processing; Relation (database); Machine translation; Information retrieval; Artificial intelligence; Relationship extraction; Resource (disambiguation); Order (exchange); Bilingual dictionary; Information extraction; Data mining","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001481726,0.001063443,0.0007031789,0.005583599,0.002048007,0.001385903,0.0008126577,0.001219742,0.002117818],"category_scores_gemma":[0.007838533,0.0005193551,0.0006119533,0.004424684,0.0006242662,0.001678563,0.002029283,0.0006872326,0.002177813],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005624981,"about_ca_system_score_gemma":0.001708162,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006789675,"about_ca_topic_score_gemma":0.01433119,"domain_scores_codex":[0.9984167,0.0003407319,0.0001729108,0.0005948298,0.0003919594,0.00008295466],"domain_scores_gemma":[0.9955952,0.002039075,0.0004340641,0.0005698987,0.001073679,0.000288108],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.000682818,0.0006879503,0.02287357,0.002501244,0.0004653698,0.003916318,0.004285162,0.005479425,0.132628,0.002639845,0.03671784,0.7871224],"study_design_scores_gemma":[0.0004915125,0.001108807,0.09793739,0.0003288292,0.0008249517,0.01625491,0.01088341,0.3009167,0.3090288,0.01928317,0.2425143,0.000427232],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.573408,0.007271396,0.3612464,0.001458995,0.0005973025,0.001838272,0.01507901,0.024102,0.01499867],"genre_scores_gemma":[0.3606927,0.001029764,0.6087334,0.0001801365,0.0002176814,0.0004428834,0.02241112,0.0006968883,0.005595502],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.006789675,"threshold_uncertainty_score":0.01350033,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02187158583543474,"score_gpt":0.3141396691065393,"score_spread":0.2922680832711045,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}