{"id":"W1987756646","doi":"10.1007/s10994-009-5151-5","title":"A co-classification approach to learning from multilingual corpora","year":2009,"lang":"en","type":"article","venue":"Machine Learning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Categorization; Artificial intelligence; Boosting (machine learning); Regularization (linguistics); Natural language processing; Benchmarking; Consistency (knowledge bases); Text categorization; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008427502,0.001671309,0.002326551,0.01119521,0.003468176,0.00528183,0.004677598,0.003267847,0.006261506],"category_scores_gemma":[0.02449719,0.0008943988,0.002709803,0.01334511,0.001580038,0.008500831,0.005624667,0.004708035,0.00577611],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001397559,"about_ca_system_score_gemma":0.003358972,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007018395,"about_ca_topic_score_gemma":0.01559406,"domain_scores_codex":[0.989226,0.003962787,0.00101834,0.002687758,0.002574493,0.0005306366],"domain_scores_gemma":[0.971939,0.0132691,0.0008657626,0.006267267,0.007112603,0.0005463818],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000357585,0.000728891,0.004864538,0.0004560324,0.0005209795,0.0004191755,0.0005801214,0.009186373,0.006530983,0.02154652,0.03035346,0.9244554],"study_design_scores_gemma":[0.0001187172,0.0002860592,0.004562896,0.0002987058,0.0007444952,0.001331131,0.0009359181,0.7817575,0.0228662,0.1128578,0.07400073,0.0002397501],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0080079,0.001757942,0.979044,0.0007989653,0.0005245376,0.0003084204,0.001073805,0.003209762,0.005274647],"genre_scores_gemma":[0.130629,0.001422585,0.8379741,0.0006938911,0.0008711594,0.001130751,0.008809052,0.0009248661,0.01754447],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01119521,"threshold_uncertainty_score":0.04456943,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02645328759020278,"score_gpt":0.2968939080229041,"score_spread":0.2704406204327013,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}