{"id":"W7116666449","doi":"10.1007/978-3-032-13509-4_25","title":"LinguaMark: Do Multimodal Models Speak Fairly? A Benchmark-Based Evaluation","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in social networks","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Vector Institute","funders":"","keywords":"Benchmark (surveying); Generalization; Key (lock); Code (set theory); Language model; Computational linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01885622,0.003577142,0.002487819,0.002438535,0.00192932,0.004978959,0.003805308,0.004794127,0.02129394],"category_scores_gemma":[0.06544055,0.0007290283,0.001514779,0.001556852,0.001467159,0.00840307,0.005393324,0.00341562,0.009721454],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002103599,"about_ca_system_score_gemma":0.001597624,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009004973,"about_ca_topic_score_gemma":0.00979643,"domain_scores_codex":[0.9810622,0.0136525,0.0008359187,0.001531706,0.002326621,0.0005909867],"domain_scores_gemma":[0.9593594,0.03181748,0.0006092917,0.003780046,0.003161455,0.001272173],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.01670851,0.005525869,0.01398137,0.004101159,0.002875925,0.000625041,0.00125322,0.1225135,0.008552889,0.01054022,0.2480314,0.565291],"study_design_scores_gemma":[0.002520382,0.003820951,0.01046286,0.0006266001,0.001139239,0.0006391262,0.001915015,0.9163287,0.01213775,0.0210749,0.02901362,0.0003209331],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6053987,0.02317639,0.1825778,0.008643115,0.004461312,0.003083982,0.03329622,0.03844627,0.1009163],"genre_scores_gemma":[0.8062488,0.002381169,0.09690236,0.001799817,0.0006885653,0.002062459,0.06326433,0.004318848,0.02233375],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02129394,"threshold_uncertainty_score":0.09972239,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01698227311453109,"score_gpt":0.3057661589263878,"score_spread":0.2887838858118567,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}