{"id":"W4386156233","doi":"10.32920/24034104","title":"Predicting Urban Functional Zones with Twitter Data Using the Space-Time Scan Statistics Method and the Random Forest Classifier","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Human Mobility and Location-Based Analysis","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Random forest; Classifier (UML); Cluster analysis; Segmentation; Pattern recognition (psychology); Computer science; Artificial intelligence; Precision and recall; Geography; Data mining; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001873221,0.0008489249,0.0007256693,0.0053147,0.0005919464,0.0009997021,0.0006552123,0.0007722588,0.00110092],"category_scores_gemma":[0.003737023,0.0002232009,0.0013534,0.002369189,0.0003145693,0.001100797,0.0005412324,0.0006034971,0.001131969],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000697667,"about_ca_system_score_gemma":0.0007663238,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01824693,"about_ca_topic_score_gemma":0.02058434,"domain_scores_codex":[0.9992791,0.0001703599,0.00005857239,0.000209117,0.0001605529,0.0001223229],"domain_scores_gemma":[0.9982592,0.0009646032,0.0001936896,0.0001377475,0.000359412,0.00008531513],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009584941,0.0005994468,0.2810077,0.0002309874,0.0003553321,0.0005591707,0.0005251992,0.2240229,0.01084351,0.002718066,0.01089629,0.4672829],"study_design_scores_gemma":[0.000013342,0.00007899685,0.02978271,0.00003133683,0.00003484824,0.0001075344,0.0003628852,0.9642157,0.002460778,0.001777085,0.001114096,0.00002071325],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7057557,0.0004065392,0.2829238,0.0004188641,0.0001083737,0.000310627,0.005468285,0.002292299,0.002315567],"genre_scores_gemma":[0.8827378,0.0001722104,0.109439,0.00003143419,0.00006253294,0.0001654126,0.006173726,0.00006517769,0.001152664],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01824693,"threshold_uncertainty_score":0.03628147,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1296583238809215,"score_gpt":0.3602088202276374,"score_spread":0.2305504963467159,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}