{"id":"W4412353994","doi":"10.2139/ssrn.5348737","title":"Land Cover Classification and Change Detection Using Large Language Models: A Novel Benchmark Study for Remote Sensing Applications","year":2025,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Benchmark (surveying); Land cover; Cover (algebra); Change detection; Remote sensing; Computer science; Data mining; Natural language processing; Artificial intelligence; Land use; Geography; Cartography; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004253,0.001476448,0.001119874,0.001975842,0.0009645871,0.00250522,0.00182465,0.002171824,0.00155553],"category_scores_gemma":[0.01293959,0.0002771972,0.001093978,0.002420607,0.000910439,0.002987483,0.001379857,0.001397836,0.0007466894],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001131977,"about_ca_system_score_gemma":0.001071371,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0154533,"about_ca_topic_score_gemma":0.01419102,"domain_scores_codex":[0.9977993,0.0009418761,0.0001659379,0.0005281021,0.0003929824,0.0001716841],"domain_scores_gemma":[0.9900702,0.006816752,0.0004985057,0.001370652,0.0009659151,0.0002780159],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002571415,0.003415741,0.05576513,0.0009589516,0.0008270594,0.0008439024,0.0003898875,0.5214036,0.01181494,0.007535961,0.01959519,0.3748782],"study_design_scores_gemma":[0.00007021798,0.0003323339,0.007256706,0.00001467789,0.00007242339,0.0001559816,0.0001880748,0.9815627,0.00346454,0.005383828,0.001471849,0.00002667435],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8646558,0.003234589,0.1143897,0.00152175,0.000289372,0.0003552615,0.007949292,0.001894087,0.005710176],"genre_scores_gemma":[0.9261455,0.000486569,0.05945997,0.0001902577,0.0002314488,0.0001552579,0.01089893,0.0001759489,0.002256267],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0154533,"threshold_uncertainty_score":0.03072673,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05303684395705223,"score_gpt":0.3159063533926723,"score_spread":0.26286950943562,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}