{"id":"W7082280113","doi":"10.48448/w2qw-q160","title":"Evaluating Visual and Cultural Interpretation: The K-Viscuit Benchmark with Human-VLM Collaboration","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Benchmark (surveying); Process (computing); Key (lock); Set (abstract data type); Baseline (sea)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00714311,0.002750104,0.0008675064,0.003815917,0.002015251,0.003885666,0.004251077,0.00417232,0.009324806],"category_scores_gemma":[0.03827357,0.0004191652,0.001535125,0.003059372,0.001715496,0.005687956,0.00651112,0.002531205,0.006687163],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002842358,"about_ca_system_score_gemma":0.002300015,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02077534,"about_ca_topic_score_gemma":0.03968086,"domain_scores_codex":[0.9882609,0.006437332,0.0009118253,0.002407195,0.001576408,0.0004062569],"domain_scores_gemma":[0.980535,0.01021076,0.0006944607,0.005035697,0.002740938,0.0007832598],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00199828,0.002646349,0.02583794,0.005463615,0.0006284463,0.0009783887,0.00528168,0.0338789,0.009379222,0.0206394,0.6123749,0.2808929],"study_design_scores_gemma":[0.001300756,0.00111592,0.03527149,0.001799213,0.0002916096,0.001671909,0.01332497,0.3059315,0.02981666,0.05598107,0.5530188,0.0004760789],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.3818346,0.008869098,0.1085166,0.005992522,0.001769339,0.004142978,0.3164625,0.05788294,0.1145293],"genre_scores_gemma":[0.3115813,0.0005650389,0.1694033,0.001809528,0.0001208753,0.00174719,0.5039816,0.002021848,0.008769345],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.02077534,"threshold_uncertainty_score":0.04130882,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02122261821966825,"score_gpt":0.3482990247563129,"score_spread":0.3270764065366447,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}