{"id":"W4283805152","doi":"10.1609/aaai.v36i2.20123","title":"Improving Zero-Shot Phrase Grounding via Reasoning on External Knowledge and Spatial Relations","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Phrase; Computer science; Noun phrase; Artificial intelligence; Ground; Visual reasoning; Natural language processing; Modal; Noun; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001005727,0.002204338,0.001452545,0.002283031,0.0009323304,0.001688768,0.003817468,0.002027213,0.006272289],"category_scores_gemma":[0.004757592,0.0006029892,0.001485666,0.001775621,0.001553428,0.007484565,0.003992333,0.002576002,0.002341981],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00125705,"about_ca_system_score_gemma":0.00192556,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01430443,"about_ca_topic_score_gemma":0.02996402,"domain_scores_codex":[0.998821,0.000138758,0.0000546286,0.0005042074,0.0003390804,0.0001422554],"domain_scores_gemma":[0.9982109,0.0008473943,0.0001234564,0.0004909037,0.0002466405,0.00008074033],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004200774,0.0005517828,0.003788483,0.0005608319,0.0001653922,0.0006319305,0.0005530936,0.05922835,0.03085542,0.01929318,0.03169161,0.8522598],"study_design_scores_gemma":[0.0001004601,0.0002667331,0.00166379,0.00009013656,0.000151827,0.0003974456,0.0005674596,0.8685213,0.03571849,0.08109743,0.01136808,0.00005681399],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0670578,0.001368768,0.8938745,0.0006804366,0.0001709938,0.0003081592,0.003071994,0.02642512,0.007042196],"genre_scores_gemma":[0.444185,0.0006731374,0.5286528,0.0008774102,0.0001132454,0.0001482792,0.01652577,0.001090347,0.007733941],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01430443,"threshold_uncertainty_score":0.02844232,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04921381018470201,"score_gpt":0.305775941823595,"score_spread":0.256562131638893,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}