{"id":"W2150417504","doi":"","title":"Creating Robust Supervised Classifiers via Web-Scale N-Gram Data","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"n-gram; Bracketing (phenomenology); Computer science; Artificial intelligence; Scale (ratio); Natural language processing; Gram; Noun; Verb; Speech recognition; Language model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004550943,0.0001537229,0.0001516358,0.00009668343,0.0001851713,0.0004205184,0.00373742,0.0001425425,0.0001061291],"category_scores_gemma":[0.0001388537,0.0001274622,0.00003473292,0.0004018552,0.00007738612,0.001393986,0.001413331,0.0005015,0.00004178928],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001603749,"about_ca_system_score_gemma":0.0000945538,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001685243,"about_ca_topic_score_gemma":0.0003874004,"domain_scores_codex":[0.9985163,0.00003246424,0.0002114521,0.0006167734,0.0002889211,0.0003341125],"domain_scores_gemma":[0.9976087,0.00009016387,0.0000666874,0.002015208,0.00009073388,0.0001285007],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000006537291,0.0001310074,0.001850908,0.00004797773,0.00001953707,0.00003459027,0.0005788869,0.000003743307,0.3204171,0.03940712,0.01557298,0.6219296],"study_design_scores_gemma":[0.0002792088,0.00004309315,0.0002358096,0.00002808507,0.000009949183,0.00005668964,0.000042953,0.9366082,0.04606698,0.01127179,0.004887607,0.000469646],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.006212481,0.0001323494,0.9792535,0.001408193,0.0003969398,0.0001472855,0.00000524262,0.002119343,0.01032471],"genre_scores_gemma":[0.2527959,0.00000265649,0.746145,0.0004439974,0.00009389305,0.000005946169,0.0000174046,0.00001151343,0.0004837245],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9366044,"threshold_uncertainty_score":0.6945118,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03026958532982212,"score_gpt":0.2801285670039378,"score_spread":0.2498589816741157,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}