{"id":"W7116933240","doi":"10.1109/etncc66224.2025.11299754","title":"Automatic Image Tagging and Captioning Using Transformer-Based Vision-Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Trinity College","funders":"","keywords":"Closed captioning; Transformer; Encoder; Feature extraction; Natural language; Visualization","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007788564,0.0008911013,0.0005355283,0.001207478,0.0003539403,0.001544241,0.001449273,0.0007594849,0.003008737],"category_scores_gemma":[0.002487146,0.0004743788,0.001504765,0.0006946626,0.0006348369,0.002494309,0.0009701239,0.001511334,0.002946114],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001351129,"about_ca_system_score_gemma":0.000972783,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007056862,"about_ca_topic_score_gemma":0.00869004,"domain_scores_codex":[0.9995454,0.00007957207,0.00003069304,0.0001684471,0.000117851,0.0000579652],"domain_scores_gemma":[0.9991097,0.0003489063,0.00008237502,0.0001358995,0.0002727484,0.00005032911],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004610935,0.000296013,0.002152595,0.0003419048,0.0001170121,0.0003661319,0.0004923398,0.17574,0.06004437,0.02402549,0.01406252,0.7219006],"study_design_scores_gemma":[0.00001195658,0.00004426083,0.0002531705,0.00001439828,0.00002184731,0.0001016233,0.0000631333,0.9696023,0.01852413,0.008087049,0.003252475,0.00002358647],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01542605,0.0002241523,0.9736302,0.0001964599,0.00008989967,0.0001282343,0.0004067707,0.007634371,0.002263834],"genre_scores_gemma":[0.4414716,0.0005778835,0.545323,0.0003892821,0.00008320162,0.000239114,0.003256472,0.0009618188,0.007697685],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007056862,"threshold_uncertainty_score":0.01403159,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01270002409126329,"score_gpt":0.322999772445022,"score_spread":0.3102997483537587,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}