{"id":"W4402684225","doi":"10.18653/v1/2024.findings-acl.189","title":"Concept-Best-Matching: Evaluating Compositionality In Emergent Communication","year":2024,"lang":"en","type":"article","venue":"","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Azrieli Foundation; Open Philanthropy Project","keywords":"Principle of compositionality; Computer science; Matching (statistics); Natural language processing; Artificial intelligence; Mathematics; Statistics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01645503,0.001536214,0.001432128,0.004176366,0.001254657,0.002377211,0.001757393,0.002745453,0.003297197],"category_scores_gemma":[0.07358398,0.0002970627,0.00142423,0.002424801,0.001643968,0.005193683,0.003440479,0.001571311,0.0006408903],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001403834,"about_ca_system_score_gemma":0.001874946,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00220078,"about_ca_topic_score_gemma":0.002199171,"domain_scores_codex":[0.9897972,0.005333784,0.0009227991,0.001294368,0.002253133,0.000398719],"domain_scores_gemma":[0.9598151,0.03195821,0.002296672,0.002415331,0.002571928,0.0009427117],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.005946963,0.001632997,0.06434111,0.001983259,0.001710687,0.0004479753,0.002289738,0.2429031,0.01185455,0.02682176,0.005825507,0.6342424],"study_design_scores_gemma":[0.000241392,0.001302398,0.008747919,0.0001183471,0.0002579814,0.0002897091,0.0008283875,0.905912,0.01085979,0.06936668,0.001982827,0.00009260912],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3875531,0.001460011,0.5991298,0.0005095026,0.0002346177,0.0009836508,0.0008298343,0.001985826,0.007313742],"genre_scores_gemma":[0.774479,0.000189184,0.2223906,0.000152966,0.0000532484,0.0004883527,0.00124694,0.0002014578,0.0007982588],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01645503,"threshold_uncertainty_score":0.08702356,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07255226108626564,"score_gpt":0.3805122086814174,"score_spread":0.3079599475951518,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}