{"id":"W4378465251","doi":"10.48550/arxiv.2305.14998","title":"An Examination of the Robustness of Reference-Free Image Captioning Evaluation Metrics","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Canadian Institute for Advanced Research; Nvidia","keywords":"Closed captioning; Robustness (evolution); Computer science; Artificial intelligence; Computer vision; Image (mathematics); Natural language processing","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03319294,0.003673662,0.001486671,0.008954925,0.001420567,0.0044544,0.00272156,0.002668935,0.002575313],"category_scores_gemma":[0.1512863,0.0005265187,0.001167288,0.00400782,0.001651417,0.004975601,0.003771692,0.002574879,0.00129969],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002437448,"about_ca_system_score_gemma":0.001165451,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006681616,"about_ca_topic_score_gemma":0.005242389,"domain_scores_codex":[0.9668292,0.01470816,0.003606573,0.004947852,0.008962906,0.0009452173],"domain_scores_gemma":[0.855915,0.08720976,0.008437557,0.01715968,0.0289702,0.002307787],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.003939934,0.0008635824,0.04991807,0.004644122,0.002540346,0.000480595,0.00253617,0.08795284,0.04422666,0.006936755,0.04821432,0.7477466],"study_design_scores_gemma":[0.0003410441,0.004936293,0.1045945,0.001057287,0.0009126396,0.001825166,0.002396418,0.7087789,0.1209072,0.01596898,0.03750182,0.0007797746],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6116742,0.02032951,0.3023904,0.001625651,0.00266041,0.002250314,0.01035961,0.02241554,0.02629433],"genre_scores_gemma":[0.8257096,0.001024636,0.1531838,0.0004808865,0.000335055,0.00074876,0.01392788,0.001875975,0.002713382],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9668071,"threshold_uncertainty_score":0.1755432,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1578930036878109,"score_gpt":0.2572585749079035,"score_spread":0.09936557122009265,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}