{"id":"W19546019","doi":"10.1016/j.jocd.2009.05.001","title":"Language Identification Strategies for Cross Language Information Retrieval.","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Language identification; Identification (biology); Task (project management); Natural language; Information retrieval; Grammar; Language model; Metadata; Linguistics; World Wide Web","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0004613706,0.00009994389,0.00008157185,0.0001192029,0.0001208786,0.001325463,0.0007984735,0.0001077693,0.00002262929],"category_scores_gemma":[0.0002402274,0.00008323285,0.00004454869,0.0002626321,0.00003887458,0.004321724,0.0001022105,0.0001771024,0.0000411492],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001647436,"about_ca_system_score_gemma":0.00007183247,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004803543,"about_ca_topic_score_gemma":0.00003028126,"domain_scores_codex":[0.9991931,0.00001104665,0.0002398127,0.0001749529,0.0001975202,0.0001835105],"domain_scores_gemma":[0.9990577,0.00005785518,0.0001335879,0.0005031141,0.0002058295,0.00004193001],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00001411101,0.00001879595,0.00003459683,0.00007791794,0.000005589895,0.00000195167,0.006668489,0.0000022735,0.416924,0.514659,0.001635475,0.05995784],"study_design_scores_gemma":[0.000364224,0.00004797102,0.0005512121,0.00001182933,0.000005319034,0.00002100497,0.001234653,0.0148516,0.9261042,0.05412798,0.002345853,0.0003341641],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07944625,0.0001214871,0.9168832,0.0003121607,0.0003485646,0.0002825077,0.000008671015,0.001149516,0.001447645],"genre_scores_gemma":[0.6794077,7.660823e-7,0.3197549,0.0002297278,0.00005811414,0.00001847576,0.00002982709,0.000004712995,0.0004957645],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.5999615,"threshold_uncertainty_score":0.9997113,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.007840132938790805,"score_gpt":0.3097372937516138,"score_spread":0.301897160812823,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}