{"id":"W2925086903","doi":"10.22215/etd/2017-11995","title":"Statistical Analysis of Classification Algorithms for Predicting Socioeconomics Status of Twitter Users","year":2017,"lang":"en","type":"dissertation","venue":"","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Support vector machine; Naive Bayes classifier; Decision tree; Computer science; Machine learning; Artificial intelligence; Statistical classification; Algorithm; Data mining","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003047462,0.0001381917,0.0004286129,0.0003678649,0.0001051955,0.00009854292,0.0004734794,0.0002128228,0.00002898471],"category_scores_gemma":[0.0001695378,0.0001396531,0.0002059264,0.0001492309,0.0000359426,0.0002491478,0.00001744971,0.000113608,0.000001425371],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008139711,"about_ca_system_score_gemma":0.0001779954,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005982264,"about_ca_topic_score_gemma":0.0004007936,"domain_scores_codex":[0.9987165,0.00002901988,0.0004993071,0.0003701177,0.0001841373,0.0002008594],"domain_scores_gemma":[0.997863,0.0002966476,0.0009589799,0.000544579,0.0002746365,0.00006217186],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004530859,0.0004557388,0.1206219,0.001713135,0.01017454,0.000001551343,0.02418006,0.002082089,0.006492825,0.1022134,0.005115932,0.7264958],"study_design_scores_gemma":[0.0003083245,0.0001463083,0.3120326,0.00003211554,0.001110552,1.614675e-7,0.000711463,0.6794145,0.003596454,0.001888269,0.0005236764,0.0002356784],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1223517,0.00003095986,0.8729803,0.00002884896,0.00105198,0.000291172,0.000205266,0.0000470191,0.003012762],"genre_scores_gemma":[0.9122321,0.00005332706,0.0836138,0.00001865104,0.00008958719,0.00004979544,0.002144898,0.00002063404,0.001777198],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7898805,"threshold_uncertainty_score":0.5694889,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05587029747400021,"score_gpt":0.340885170468379,"score_spread":0.2850148729943788,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}