{"id":"W4403888851","doi":"10.1007/978-3-031-73039-9_4","title":"MEERKAT: Audio-Visual Large Language Model for Grounding in Space and Time","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Music and Audio Processing","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Audio visual; Ground; Space (punctuation); Speech recognition; Computer graphics (images); Electrical engineering; Multimedia; Operating system; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001108633,0.0004310863,0.0004777352,0.0007750274,0.0002020474,0.0008569136,0.001277719,0.000251174,0.000006814159],"category_scores_gemma":[0.00005761219,0.000391368,0.00008657059,0.0004828013,0.0002536658,0.0006501579,0.001240677,0.00058413,0.00002163502],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002086461,"about_ca_system_score_gemma":0.0003374823,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00000830971,"about_ca_topic_score_gemma":0.00003901915,"domain_scores_codex":[0.9969457,0.00001259893,0.000362821,0.001440027,0.0005282161,0.0007106115],"domain_scores_gemma":[0.9988602,0.0003036624,0.0001412312,0.0004974682,0.00007199393,0.0001254315],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001597208,0.00006262489,0.0000321569,0.000617067,0.00002947088,0.000294765,0.01618623,0.0357932,0.001577645,0.3300373,0.0002525331,0.615101],"study_design_scores_gemma":[0.0002053475,0.00004560507,0.000005963877,0.0005466932,0.000006656362,0.00002574574,5.250554e-7,0.8280802,0.000255418,0.170091,0.0003501094,0.0003867083],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0002856002,0.001717083,0.9927438,0.001149605,0.0006239702,0.0003444662,0.000008921293,0.0001430125,0.002983504],"genre_scores_gemma":[0.4171049,0.00004332472,0.5719665,0.002936165,0.0008634594,0.00002779275,0.000009606699,0.00008579683,0.00696244],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.792287,"threshold_uncertainty_score":0.9998538,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.014323516536009,"score_gpt":0.2710715640374734,"score_spread":0.2567480475014644,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}