{"id":"W2002586403","doi":"10.3115/1067807.1067851","title":"Bootstrapping statistical parsers from small datasets","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":156,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"Defense Advanced Research Projects Agency; Johns Hopkins University; National Science Foundation","keywords":"Bootstrapping (finance); Parsing; Computer science; Strapping; Domain (mathematical analysis); Artificial intelligence; Natural language processing; Statistical analysis; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01332503,0.002037203,0.001894092,0.004020429,0.001620843,0.00169182,0.003415873,0.00234103,0.003974121],"category_scores_gemma":[0.1018739,0.001752707,0.001823623,0.00371102,0.001438325,0.004728971,0.003539603,0.004605644,0.004105057],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001089177,"about_ca_system_score_gemma":0.002646616,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001575625,"about_ca_topic_score_gemma":0.00404618,"domain_scores_codex":[0.9868347,0.007621726,0.0006979769,0.002562178,0.00196109,0.0003223373],"domain_scores_gemma":[0.8693045,0.09826329,0.002632898,0.02108045,0.00785659,0.000862266],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0008395145,0.001002057,0.0070716,0.0009190144,0.0005725211,0.001252148,0.001027194,0.1203075,0.04780336,0.006629388,0.0383737,0.774202],"study_design_scores_gemma":[0.0002044567,0.0003185652,0.004195542,0.0001175381,0.000185963,0.0006089423,0.0003027159,0.8909331,0.04435293,0.03985252,0.01877534,0.0001523787],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04777188,0.0003289788,0.9247446,0.0004798641,0.0001852151,0.0006520098,0.002015985,0.02229008,0.001531362],"genre_scores_gemma":[0.1895236,0.0002310244,0.7861127,0.0005947909,0.000245576,0.00214037,0.01654129,0.003273512,0.001337194],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01332503,"threshold_uncertainty_score":0.07047039,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0253997854399811,"score_gpt":0.2756000191421251,"score_spread":0.250200233702144,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}