{"id":"W3129531092","doi":"10.1109/icdmw51313.2020.00088","title":"Graph-based Topic Extraction Using Centroid Distance of Phrase Embeddings on Healthy Aging Open-ended Survey Questions","year":2020,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Natural language processing; Phrase; Artificial intelligence; Information retrieval; Word (group theory); Graph; Centroid; Task (project management); Domain (mathematical analysis); Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001288946,0.001194428,0.0008632034,0.005025094,0.000405758,0.001274821,0.0007792714,0.001061946,0.002439491],"category_scores_gemma":[0.006373059,0.0002509821,0.001082213,0.003896064,0.0003416286,0.001829633,0.001088647,0.0008125807,0.002789462],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005703655,"about_ca_system_score_gemma":0.0007095957,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003268841,"about_ca_topic_score_gemma":0.004307342,"domain_scores_codex":[0.9986545,0.000467752,0.0001180996,0.0004367445,0.000188994,0.0001338172],"domain_scores_gemma":[0.9966589,0.002035527,0.0002957405,0.0002774991,0.0006275147,0.0001047843],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001048662,0.0002977063,0.01007312,0.0009124505,0.0002251731,0.0003444076,0.001423849,0.02896333,0.02100864,0.006829649,0.02186763,0.9070054],"study_design_scores_gemma":[0.00015007,0.000369151,0.01746739,0.0001255779,0.0001601016,0.0004675762,0.001715295,0.9257773,0.01168339,0.02419889,0.01778915,0.00009602005],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1632841,0.002458385,0.8185581,0.0005298259,0.0003173078,0.0004917372,0.005748305,0.005419096,0.003193049],"genre_scores_gemma":[0.6174008,0.001116025,0.352956,0.0001485583,0.0003323849,0.0006682075,0.02155659,0.0004297659,0.005391714],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005025094,"threshold_uncertainty_score":0.008160949,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06804492174385719,"score_gpt":0.3786420053132639,"score_spread":0.3105970835694067,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}