{"id":"W4412353994","doi":"10.2139/ssrn.5348737","title":"Land Cover Classification and Change Detection Using Large Language Models: A Novel Benchmark Study for Remote Sensing Applications","year":2025,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Benchmark (surveying); Land cover; Cover (algebra); Change detection; Remote sensing; Computer science; Data mining; Natural language processing; Artificial intelligence; Land use; Geography; Cartography; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001411983,0.0002611254,0.0002790067,0.0004435215,0.0005103523,0.0003918567,0.0006353462,0.0002565357,5.448215e-7],"category_scores_gemma":[0.00004643216,0.0002585886,0.0001034596,0.0003731295,0.00003062383,0.0004470794,0.0005099188,0.001484988,0.000001008077],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009197069,"about_ca_system_score_gemma":0.0008057004,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002018203,"about_ca_topic_score_gemma":0.0007751665,"domain_scores_codex":[0.9976562,0.00006972715,0.0004032135,0.0006303262,0.0002511425,0.0009893803],"domain_scores_gemma":[0.9985142,0.00008029774,0.0004928615,0.0006614739,0.000202775,0.0000483518],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005231748,0.0003258265,0.0003648743,0.0001391062,0.0003846939,0.000001198456,0.003134199,0.0005101375,0.002210592,0.1099397,0.000009272766,0.8829281],"study_design_scores_gemma":[0.0008330211,0.0001135432,0.0003386936,0.00006301967,0.0001032962,0.00005653877,0.002077912,0.7811908,0.00009955775,0.2143987,0.0004322318,0.0002926245],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03032641,0.002088701,0.9648691,0.0005444962,0.0002108939,0.001649691,0.00002381007,0.0002268809,0.00006002814],"genre_scores_gemma":[0.97299,0.001111314,0.02529421,0.00005820133,0.0002113409,0.00006230889,0.00001894268,0.00001750545,0.0002362175],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9426636,"threshold_uncertainty_score":0.9999866,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05303684395705223,"score_gpt":0.3159063533926723,"score_spread":0.26286950943562,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}