{"id":"W2044810048","doi":"10.1002/asi.22696","title":"Mining a multilingual association dictionary from<scp>W</scp>ikipedia for cross‐language information retrieval","year":2012,"lang":"en","type":"article","venue":"Journal of the American Society for Information Science and Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"","keywords":"Computer science; Cross-language information retrieval; Information retrieval; Variety (cybernetics); Association (psychology); Natural language processing; Graph; Artificial intelligence; Machine translation; Quality (philosophy); Query expansion; Meaning (existential); Filter (signal processing)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"bench_or_experimental","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"},{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"}],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002446741,0.0001071953,0.0001911944,0.0003274434,0.0005158802,0.0003240455,0.0009765493,0.0001123962,1.912374e-7],"category_scores_gemma":[0.006377744,0.00007506843,0.0001575006,0.001799527,0.0004437295,0.009963359,0.0002547235,0.000241499,0.000001020145],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003137364,"about_ca_system_score_gemma":0.0002955687,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000006354222,"about_ca_topic_score_gemma":2.29628e-7,"domain_scores_codex":[0.9984352,0.00001209991,0.0005079422,0.00008126711,0.0006231461,0.000340353],"domain_scores_gemma":[0.9956913,0.0005069911,0.00198997,0.0002054785,0.001537658,0.00006859844],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00009843627,0.0001197989,0.03876728,0.0001439774,0.0002194266,3.256883e-7,0.1069724,0.00005075,0.0390165,0.01898814,0.0300749,0.7655481],"study_design_scores_gemma":[0.005521106,0.001885578,0.02755383,0.0002739673,0.0002898923,0.0004269638,0.09969717,0.09842775,0.486299,0.02442406,0.2543432,0.0008575013],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7575004,0.0003565747,0.2394375,0.001650956,0.0005540353,0.0002887886,0.00002688662,0.0001438187,0.00004107668],"genre_scores_gemma":[0.6803782,0.00003417845,0.3180887,0.001339346,0.0001259201,0.00001121876,0.000003649387,0.000003206536,0.00001559203],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7646905,"threshold_uncertainty_score":0.7635216,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.008604640807155386,"score_gpt":0.2943862322322477,"score_spread":0.2857815914250924,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}