{"id":"W2396331625","doi":"","title":"Merging Different Languages in a Single Document Collection.","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Search engine indexing; Natural language processing; Information retrieval; Set (abstract data type); Artificial intelligence; Document retrieval; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008436723,0.001135936,0.001462921,0.0048282,0.002090945,0.003744556,0.001764319,0.001166114,0.005195157],"category_scores_gemma":[0.02003026,0.0008025591,0.001807932,0.006080648,0.001105881,0.007784247,0.003813376,0.001404128,0.003475351],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001594078,"about_ca_system_score_gemma":0.002059448,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003251423,"about_ca_topic_score_gemma":0.005803091,"domain_scores_codex":[0.9915181,0.00329913,0.0007810693,0.001525534,0.002452941,0.0004232108],"domain_scores_gemma":[0.9862212,0.006029605,0.0007054394,0.003055622,0.003549788,0.000438385],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002975841,0.001186906,0.01001012,0.002610665,0.0008044619,0.001192133,0.004547135,0.01214572,0.1134329,0.01132489,0.02955343,0.8102159],"study_design_scores_gemma":[0.001114252,0.002793965,0.02029547,0.0005539978,0.002633563,0.004520455,0.006901809,0.1442389,0.5105448,0.03587261,0.2697363,0.0007938091],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4282043,0.006805506,0.4944704,0.001829976,0.0007676285,0.003258238,0.01121637,0.02956761,0.02387998],"genre_scores_gemma":[0.2665535,0.0006756429,0.6958117,0.0005933709,0.0001684882,0.0008120015,0.02214352,0.002587998,0.01065384],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008436723,"threshold_uncertainty_score":0.04461819,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01311184511933987,"score_gpt":0.2545793652213054,"score_spread":0.2414675201019655,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}