{"id":"W4414317082","doi":"10.20944/preprints202509.1607.v1","title":"Capturing Narrative Semantics from Captions for Relational Scene Abstraction","year":2025,"lang":"en","type":"preprint","venue":"Preprints.org","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Interpretability; Scene graph; Closed captioning; Semantics (computer science); Graph; Scalability; Natural language; Exploit; Abstraction","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005532259,0.001822283,0.0004940072,0.001410052,0.0005061019,0.001632132,0.001929851,0.001071316,0.007365675],"category_scores_gemma":[0.004015193,0.0007900497,0.001695492,0.000736755,0.0008783362,0.003331364,0.001895466,0.001951272,0.002320816],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009372485,"about_ca_system_score_gemma":0.00062238,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003799542,"about_ca_topic_score_gemma":0.009109168,"domain_scores_codex":[0.9994635,0.0001422187,0.00002309398,0.0002309303,0.0001023532,0.000037858],"domain_scores_gemma":[0.9989643,0.0004113752,0.0001092114,0.0003214538,0.00013919,0.00005439215],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004023778,0.0002720906,0.00420371,0.001298254,0.0002039506,0.001127281,0.00212008,0.2385017,0.05889883,0.07260773,0.04366132,0.5767028],"study_design_scores_gemma":[0.00002293371,0.00007380395,0.0009005655,0.00006909607,0.0000515861,0.0002541487,0.0002727286,0.9099052,0.0165245,0.05034253,0.02154873,0.00003416793],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02749116,0.0005046249,0.94313,0.0003638903,0.00009559905,0.0003364968,0.003019937,0.01986781,0.005190438],"genre_scores_gemma":[0.3573041,0.0005629814,0.6198247,0.0003142162,0.0000773837,0.000328329,0.01465016,0.002284465,0.004653628],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007365675,"threshold_uncertainty_score":0.02464062,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1403948854982185,"score_gpt":0.3842923063313395,"score_spread":0.243897420833121,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}