{"id":"W4401468895","doi":"10.1016/j.geomat.2024.100004","title":"Understanding the impact of geotagging on location inference models for accurate generalization to non-geotagged datasets","year":2024,"lang":"en","type":"article","venue":"GEOMATICA","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Geotagging; Inference; Generalization; Computer science; Geography; Data mining; Artificial intelligence; Information retrieval; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03683135,0.001382412,0.001118202,0.001210458,0.0009569391,0.003658277,0.002150475,0.002032124,0.002087865],"category_scores_gemma":[0.18209,0.0009261024,0.001785427,0.001830986,0.001579391,0.01093226,0.002631855,0.004212616,0.001188825],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001646514,"about_ca_system_score_gemma":0.002201363,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01699564,"about_ca_topic_score_gemma":0.01896026,"domain_scores_codex":[0.989099,0.00689249,0.0008421881,0.001999084,0.0007318753,0.0004352815],"domain_scores_gemma":[0.8442211,0.1312756,0.004197785,0.01547457,0.003933229,0.0008977361],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001335567,0.0006544042,0.2460397,0.000915442,0.001667185,0.0004749203,0.001437485,0.528285,0.00709964,0.01023211,0.01106761,0.1907909],"study_design_scores_gemma":[0.0000476845,0.000263925,0.0282091,0.0001375209,0.0002035395,0.0001206797,0.0005457049,0.9443722,0.003467071,0.02009193,0.002481594,0.00005900775],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5104802,0.002515763,0.4616811,0.008639154,0.0007066801,0.0006408946,0.005608332,0.003176022,0.006551754],"genre_scores_gemma":[0.9215587,0.0004524577,0.07140688,0.0007183703,0.0002011468,0.0002738105,0.004589555,0.0001624628,0.0006366652],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03683135,"threshold_uncertainty_score":0.1947851,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1157011618042064,"score_gpt":0.3475959528579116,"score_spread":0.2318947910537051,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}