{"id":"W4251596819","doi":"10.1007/978-3-540-78646-7_22","title":"Automatic Extraction of Domain-Specific Stopwords from Labeled Documents","year":2008,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Classifier (UML); Information loss; Filter (signal processing); Data mining; Set (abstract data type); Artificial intelligence; Information retrieval","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007644778,0.002187424,0.001986004,0.006268492,0.001396435,0.00250564,0.001379548,0.001840924,0.00230274],"category_scores_gemma":[0.003164721,0.0007532107,0.001246348,0.004833257,0.0004474839,0.00154096,0.0009874395,0.001640988,0.007610884],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006389024,"about_ca_system_score_gemma":0.00229727,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002476505,"about_ca_topic_score_gemma":0.005310946,"domain_scores_codex":[0.998919,0.0001417362,0.0001737856,0.0002539677,0.0003906132,0.0001210055],"domain_scores_gemma":[0.9953095,0.001508193,0.0004611314,0.0003518362,0.002210665,0.0001587206],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007764255,0.00041706,0.006500702,0.003011819,0.0002641183,0.002741187,0.0006736444,0.002585668,0.2917601,0.002570155,0.04176809,0.646931],"study_design_scores_gemma":[0.0002522324,0.000784207,0.0280547,0.0008588358,0.001352571,0.007786584,0.001984518,0.2016178,0.6168149,0.01039246,0.1298112,0.0002899544],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2388124,0.0090813,0.6758901,0.001081767,0.001437336,0.001370654,0.0222636,0.03889545,0.01116738],"genre_scores_gemma":[0.2378403,0.003924037,0.658913,0.0005011,0.0004083557,0.0006231207,0.08146454,0.002116113,0.01420954],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006268492,"threshold_uncertainty_score":0.007703424,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01686015710638942,"score_gpt":0.2507353239094569,"score_spread":0.2338751668030674,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}