{"id":"W4402464101","doi":"10.11159/mvml24.115","title":"Development of a Multimodal Framework for Deepfake Detection: Combining Visual and Audio Analysis","year":2024,"lang":"en","type":"article","venue":"Proceedings of the World Congress on Electrical Engineering and Computer Systems and Science","topic":"Digital Media Forensic Detection","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Audio visual; Audio analyzer; Speech recognition; Artificial intelligence; Human–computer interaction; Multimedia; Audio signal processing; Audio signal; Speech coding","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006890719,0.001812781,0.0009875845,0.001858454,0.0003792043,0.0009855626,0.001533215,0.0012146,0.003842055],"category_scores_gemma":[0.001975736,0.0003414887,0.0008860792,0.0007589582,0.0004054311,0.001519913,0.001951213,0.001621106,0.001980089],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006863575,"about_ca_system_score_gemma":0.0009051498,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005449928,"about_ca_topic_score_gemma":0.00956433,"domain_scores_codex":[0.9994519,0.00007845287,0.00002507765,0.0001577017,0.0001594333,0.0001274699],"domain_scores_gemma":[0.9995371,0.0001050808,0.00005046626,0.00009933809,0.0001602831,0.00004771733],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004646678,0.0003861368,0.003992135,0.0002749663,0.0001959946,0.0002980154,0.00009277594,0.0527626,0.04183612,0.003098564,0.01836382,0.8782343],"study_design_scores_gemma":[0.00002908684,0.0001773203,0.002556674,0.00006155718,0.00005636863,0.0003061129,0.00006698826,0.9543015,0.02953534,0.005003053,0.007868168,0.00003777851],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06604612,0.002513072,0.9028481,0.0007999098,0.0003494614,0.0003883069,0.00315352,0.01472766,0.009173915],"genre_scores_gemma":[0.5712798,0.001023615,0.4034652,0.0007173921,0.0001701684,0.0004448089,0.007265905,0.0003244697,0.01530865],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005449928,"threshold_uncertainty_score":0.01285297,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.006963104688251368,"score_gpt":0.2269524833259458,"score_spread":0.2199893786376944,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}