{"id":"W2141325213","doi":"10.1109/icsmc.2001.969854","title":"Filtering noisy parallel corpora of web pages","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Parallel corpora; Translation (biology); Machine translation; Natural language processing; Artificial intelligence; Information retrieval; Web page; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0000858442,0.00007959293,0.0001096103,0.00006963961,0.00002938214,0.00005361313,0.000712321,0.00004033108,0.0001203356],"category_scores_gemma":[0.00002612941,0.00006259688,0.00003481477,0.000211349,0.00002999817,0.0003339147,0.0002252499,0.0000774576,0.00002637724],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00000949789,"about_ca_system_score_gemma":0.00000694202,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001015337,"about_ca_topic_score_gemma":0.000003305895,"domain_scores_codex":[0.9993649,0.00001488517,0.0001472182,0.00017947,0.0001540075,0.0001395561],"domain_scores_gemma":[0.9994514,0.00003184182,0.00007779682,0.0003573619,0.00004668297,0.00003490758],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00000609651,0.0002227195,0.001260174,0.0001920624,0.00003310956,0.0001423001,0.001303356,0.00002746076,0.215903,0.32038,0.06624208,0.3942876],"study_design_scores_gemma":[0.000773928,0.000337077,0.0005337634,0.0002345452,0.00001192468,0.0001440294,0.00003797608,0.3515687,0.5485116,0.08679307,0.01005352,0.0009998915],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00382925,0.002350584,0.9775296,0.001143806,0.00009764737,0.0000895987,9.240181e-7,0.001004475,0.01395406],"genre_scores_gemma":[0.4775868,0.00003150547,0.5214416,0.0001265203,0.00001080717,0.000002601723,1.865114e-7,0.000003067094,0.0007969569],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.4737575,"threshold_uncertainty_score":0.2552626,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0297197218768101,"score_gpt":0.243206128264564,"score_spread":0.2134864063877539,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}