{"id":"W2509251152","doi":"10.18653/v1/w16-2370","title":"BAD LUC$@$WMT 2016: a Bilingual Document Alignment Platform Based on Lucene","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"Fonds de recherche du Québec – Nature et technologies","keywords":"Computer science; Task (project management); Information retrieval; Natural language processing; Heuristic; Artificial intelligence; World Wide Web","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004233758,0.0002200166,0.0001681741,0.0001661938,0.00009447215,0.000155314,0.001057891,0.00009127655,0.0002064561],"category_scores_gemma":[0.00007591498,0.0001184854,0.00007738546,0.000219349,0.0000434946,0.0005267194,0.0002874414,0.00009742068,0.0001848683],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002274098,"about_ca_system_score_gemma":0.0001887451,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003759468,"about_ca_topic_score_gemma":0.000008377297,"domain_scores_codex":[0.9982006,0.00002905246,0.0002646306,0.000545344,0.0005516623,0.000408736],"domain_scores_gemma":[0.9986851,0.0001665641,0.00009618183,0.0008409666,0.00007507973,0.0001360641],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00006389109,0.0002311852,0.00009900523,0.0000253583,0.00002180393,0.0001106626,0.0002333167,0.00000854331,0.02654379,0.2229497,0.01288156,0.7368311],"study_design_scores_gemma":[0.001066814,0.0005581602,0.00004958856,0.0003414844,0.000007103974,0.00001802176,0.000009273444,0.005555344,0.9162772,0.06329355,0.01224406,0.0005793877],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.003498554,0.0002641999,0.9827331,0.005438367,0.0003168941,0.0002999462,0.000002981345,0.001430268,0.006015685],"genre_scores_gemma":[0.5680816,0.00001033767,0.426456,0.002242379,0.00007157828,0.00002550044,0.000001178182,0.00001348096,0.003097956],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8897334,"threshold_uncertainty_score":0.4831695,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01097485922180609,"score_gpt":0.267082588178196,"score_spread":0.2561077289563899,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}