{"id":"W2510939506","doi":"10.18653/v1/w16-0417","title":"Semi-supervised and unsupervised categorization of posts in Web discussion forums using part-of-speech information and minimal features","year":2016,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Categorization; Artificial intelligence; Cluster analysis; Hidden Markov model; Identification (biology); Probabilistic logic; Topic model; Natural language processing; Machine learning; Unsupervised learning; Information retrieval","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001799364,0.00007917591,0.0001269783,0.0001464207,0.00002901218,0.00002564104,0.0001281474,0.00006271976,0.0000040879],"category_scores_gemma":[0.0000369317,0.00004317382,0.00001390137,0.0001622833,0.00003010884,0.001259603,0.0001456976,0.00003105807,3.627742e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001628489,"about_ca_system_score_gemma":0.00004564926,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000709011,"about_ca_topic_score_gemma":0.0000417938,"domain_scores_codex":[0.9992592,0.00003168033,0.0002964881,0.0001383506,0.0001522378,0.0001219874],"domain_scores_gemma":[0.9996036,0.00003931225,0.00008282188,0.0001730595,0.00006248983,0.00003868095],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004803762,0.00003355068,0.05991837,0.0001621003,0.000009630643,0.000001369673,0.004815818,0.0001959526,0.1972174,0.01353608,0.00005382097,0.7240079],"study_design_scores_gemma":[0.002780443,0.0001533797,0.03118091,0.0003544227,0.00001130585,0.00002460904,0.0005347723,0.8903686,0.07198754,0.002014974,0.0002646787,0.0003243433],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8337263,0.00004630471,0.1652101,0.0007059325,0.00005701232,0.0001324004,0.000006161154,0.00001895443,0.00009679187],"genre_scores_gemma":[0.9782513,0.00004084967,0.02163794,0.00003106951,0.00001050986,0.000001639842,0.000002749153,0.000002231528,0.00002167743],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8901727,"threshold_uncertainty_score":0.1760577,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0154227995785618,"score_gpt":0.234658583576162,"score_spread":0.2192357839976002,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}