{"id":"W763189797","doi":"10.3758/s13428-015-0614-z","title":"Performance impact of stop lists and morphological decomposition on word–word corpus-based semantic space models","year":2015,"lang":"en","type":"article","venue":"Behavior Research Methods","topic":"Topic Modeling","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Word (group theory); Word lists by frequency; Space (punctuation); Semantics (computer science); Class (philosophy); Text corpus; Linguistics; Sentence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008810797,0.002029446,0.002011898,0.002142156,0.00109664,0.003171943,0.001804483,0.002035659,0.005615626],"category_scores_gemma":[0.02505166,0.0007043006,0.001808107,0.002076844,0.0004642683,0.004878114,0.002028257,0.002686111,0.003521552],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001468977,"about_ca_system_score_gemma":0.00398178,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.040212,"about_ca_topic_score_gemma":0.03300215,"domain_scores_codex":[0.9963956,0.002127101,0.0003491135,0.0006211416,0.0002907586,0.0002162208],"domain_scores_gemma":[0.9738045,0.02205531,0.0003554442,0.001235549,0.001821256,0.0007279076],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.006060257,0.001132292,0.01457409,0.0006121611,0.001375525,0.0001225724,0.0004572384,0.2580611,0.005821026,0.003242376,0.01402256,0.6945188],"study_design_scores_gemma":[0.00003656372,0.00007593806,0.0008584157,0.0000180946,0.00009117262,0.00001941547,0.00013108,0.9958188,0.001476109,0.001070133,0.0003871422,0.00001719884],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.6010895,0.005763468,0.3480953,0.00254804,0.0008671596,0.0002774936,0.004454814,0.03117754,0.005726552],"genre_scores_gemma":[0.8126616,0.001124166,0.1680396,0.0003145479,0.0001755282,0.0002670623,0.0122534,0.001498013,0.003666069],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.040212,"threshold_uncertainty_score":0.07995588,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4101321547628593,"score_gpt":0.5636734972443137,"score_spread":0.1535413424814544,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}