{"id":"W4404481617","doi":"10.1145/3687123.3698286","title":"Evaluation of Code LLMs on Geospatial Code Generation","year":2024,"lang":"en","type":"article","venue":"","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Geospatial analysis; Computer science; Code (set theory); Computer security; Programming language; Database; Geography; Remote sensing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008122085,0.002411891,0.0008270532,0.00366249,0.001209317,0.002411115,0.004099082,0.002289934,0.002608499],"category_scores_gemma":[0.04918696,0.0007502664,0.001848299,0.003432687,0.001824366,0.004135869,0.003308491,0.002970869,0.001953415],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003775978,"about_ca_system_score_gemma":0.004093945,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01599453,"about_ca_topic_score_gemma":0.01999109,"domain_scores_codex":[0.987121,0.003926301,0.001553305,0.002204122,0.004404073,0.0007911666],"domain_scores_gemma":[0.96074,0.02144458,0.001939729,0.0082277,0.006277933,0.001370063],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.003107602,0.00345684,0.04564498,0.006834745,0.0007155355,0.0007066781,0.001748778,0.2623574,0.01895316,0.01154404,0.1611156,0.4838147],"study_design_scores_gemma":[0.0009320031,0.001736152,0.02046035,0.0008477063,0.0002977005,0.0005231345,0.0009658471,0.838587,0.04330384,0.01106793,0.08108109,0.0001973017],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7311671,0.008063058,0.0847601,0.003919384,0.001585668,0.002024993,0.0303352,0.1082874,0.02985703],"genre_scores_gemma":[0.6112946,0.002132657,0.2357455,0.001616516,0.000171533,0.001525901,0.1325313,0.009218853,0.005763105],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01599453,"threshold_uncertainty_score":0.04295421,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1167048333718861,"score_gpt":0.3446862826511968,"score_spread":0.2279814492793107,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}