{"id":"W3023453969","doi":"10.1007/978-3-030-47358-7_19","title":"Using Topic Modelling to Improve Prediction of Financial Report Commentary Classes","year":2020,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Latent Dirichlet allocation; Class (philosophy); Task (project management); Feature selection; Selection (genetic algorithm); Artificial intelligence; Machine learning; Data mining; Feature (linguistics); Finance; Topic model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003326398,0.001133782,0.0009005132,0.004098197,0.0004994915,0.002224682,0.000973158,0.001460823,0.003759841],"category_scores_gemma":[0.01488737,0.0003676807,0.001480627,0.002736423,0.0001742633,0.001779342,0.0007264304,0.002152965,0.004505798],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008596048,"about_ca_system_score_gemma":0.0007776816,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01067744,"about_ca_topic_score_gemma":0.01043232,"domain_scores_codex":[0.9986266,0.0005757197,0.000111898,0.0003321882,0.0002187339,0.0001349532],"domain_scores_gemma":[0.9831638,0.01351144,0.0009042728,0.0004617838,0.001498944,0.0004598769],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003701089,0.001718048,0.1578889,0.000842862,0.0007564126,0.0003115476,0.000656028,0.0757333,0.01203163,0.001974149,0.08261554,0.6617705],"study_design_scores_gemma":[0.0001082765,0.0003353349,0.05079397,0.00008879005,0.0003487232,0.0001074979,0.0002715203,0.930118,0.005760497,0.003203334,0.008803004,0.00006107242],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.7660259,0.01058119,0.1650612,0.003629134,0.002310241,0.0003484073,0.026867,0.01024461,0.01493229],"genre_scores_gemma":[0.9295074,0.001258804,0.03148069,0.0002351574,0.001567412,0.0002181139,0.02723548,0.0003321273,0.00816484],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.01067744,"threshold_uncertainty_score":0.02123058,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03398848257873614,"score_gpt":0.2817074990245209,"score_spread":0.2477190164457847,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}