{"id":"W1589919686","doi":"","title":"Does Unlabeled Data Provably Help? Worst-case Analysis of the Sample Complexity of Semi-Supervised Learning.","year":2008,"lang":"en","type":"article","venue":"Conference on Learning Theory","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":108,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Sample complexity; Computer science; Semi-supervised learning; Homogeneous; Distribution (mathematics); Artificial intelligence; Conjecture; Sample (material); Class (philosophy); Labeled data; Supervised learning; Machine learning; Pattern recognition (psychology); Mathematics; Artificial neural network; Discrete mathematics; Combinatorics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04909202,0.002055194,0.003230908,0.001854699,0.002274829,0.006628438,0.004759783,0.004406094,0.005380261],"category_scores_gemma":[0.2437087,0.001662931,0.002425974,0.002347252,0.006966659,0.01762641,0.006678529,0.008449861,0.0006740681],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005287651,"about_ca_system_score_gemma":0.003938376,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001770247,"about_ca_topic_score_gemma":0.002176097,"domain_scores_codex":[0.9686936,0.02026018,0.001149181,0.003399631,0.004714407,0.001783043],"domain_scores_gemma":[0.5536621,0.4099721,0.008232041,0.01941536,0.00544474,0.003273695],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.002027699,0.0005622571,0.01044493,0.0009811625,0.0006080166,0.000640019,0.0008329389,0.4680459,0.002348514,0.4380042,0.01347208,0.06203231],"study_design_scores_gemma":[0.00007958779,0.000131083,0.0005890534,0.00006714119,0.00005694027,0.0001891417,0.00009221874,0.6213314,0.001031848,0.3752761,0.001127078,0.00002841321],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.07923751,0.002680105,0.8899365,0.01374921,0.0002869549,0.0003119658,0.00108923,0.0006963228,0.01201226],"genre_scores_gemma":[0.7826052,0.001618824,0.2047564,0.002528409,0.001282519,0.0008575924,0.001588077,0.0005946105,0.004168412],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.04909202,"threshold_uncertainty_score":0.2596266,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1094726326608954,"score_gpt":0.3005835196592879,"score_spread":0.1911108869983926,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}