{"id":"W6912323747","doi":"10.5281/zenodo.16729213","title":"MedTranscripts - A multimodal dataset of Spanish medical videos and time-aligned transcripts","year":2025,"lang":"en","type":"dataset","venue":"DIGITAL.CSIC (Spanish National Research Council (CSIC))","topic":"Voice and Speech Disorders","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Pronunciation; Gold standard (test); Medical record; Human interaction","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","research_integrity","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.009775899,0.0009200087,0.001736303,0.001033737,0.0003867124,0.0005061453,0.001538601,0.001309517,0.002347624],"category_scores_gemma":[0.04793722,0.0008794043,0.000439913,0.001501221,0.001702912,0.0009691005,0.0004956017,0.00201891,0.0002326757],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002354382,"about_ca_system_score_gemma":0.03573424,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001487587,"about_ca_topic_score_gemma":0.001413833,"domain_scores_codex":[0.9652891,0.00035498,0.001660288,0.001725173,0.02955126,0.001419219],"domain_scores_gemma":[0.9811818,0.00225037,0.0003230977,0.001386272,0.01346443,0.001394061],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0009343777,0.00147645,0.00002485619,0.001894233,0.0008563655,0.0004723253,0.000110467,0.000001332913,0.0003349368,0.0002645746,0.9905516,0.003078512],"study_design_scores_gemma":[0.00495481,0.0007354186,0.0001438637,0.001490702,0.0002955666,0.0001323518,0.0001742294,0.0002892338,0.0001055444,0.0009412236,0.9900752,0.0006618838],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.0004148979,0.001967011,0.00003782882,0.004032899,0.000296075,0.001949605,0.9770893,0.00008658294,0.01412578],"genre_scores_gemma":[0.003118873,0.002161716,0.0001049567,0.002430297,0.0004660125,0.0002360926,0.9877038,0.00007610447,0.00370217],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.03816132,"threshold_uncertainty_score":0.999987,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09798634213428678,"score_gpt":0.3750296788847213,"score_spread":0.2770433367504345,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}