{"id":"W4403888851","doi":"10.1007/978-3-031-73039-9_4","title":"MEERKAT: Audio-Visual Large Language Model for Grounding in Space and Time","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Music and Audio Processing","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Audio visual; Ground; Space (punctuation); Speech recognition; Computer graphics (images); Electrical engineering; Multimedia; Operating system; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007980711,0.001794053,0.001106201,0.0008020947,0.0005083491,0.002036956,0.002856518,0.001518714,0.02602671],"category_scores_gemma":[0.003719029,0.00106926,0.001945468,0.001057486,0.000548918,0.003264022,0.001603776,0.002589538,0.01577022],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00106937,"about_ca_system_score_gemma":0.001384119,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01356318,"about_ca_topic_score_gemma":0.01651645,"domain_scores_codex":[0.9993237,0.0001206472,0.00005831871,0.0002132941,0.0002186869,0.00006520849],"domain_scores_gemma":[0.9990391,0.0003920238,0.00004904101,0.0002432521,0.0002257875,0.00005087067],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001679617,0.0002279731,0.0009873934,0.0008074776,0.0002449539,0.0003226907,0.0003108573,0.1207253,0.0279334,0.04873302,0.2175692,0.5804582],"study_design_scores_gemma":[0.0001666048,0.00007254336,0.0002021281,0.00004964382,0.00006071222,0.00009885327,0.00004728994,0.8961552,0.01559168,0.0439365,0.04356364,0.00005514284],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002571374,0.0003401571,0.898835,0.0002373025,0.0001999994,0.00008462861,0.008244393,0.08727784,0.00220922],"genre_scores_gemma":[0.1076414,0.0004336393,0.8269971,0.0004872593,0.0001315901,0.0004647118,0.03201634,0.01717484,0.01465317],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02602671,"threshold_uncertainty_score":0.08706796,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.014323516536009,"score_gpt":0.2710715640374734,"score_spread":0.2567480475014644,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}