{"id":"W4406930353","doi":"10.21203/rs.3.rs-5656576/v1","title":"RETRACTED: Repurposing the Scientific Literature with Vision-Language Models","year":2025,"lang":"en","type":"preprint","venue":"Research Square","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":true,"has_abstract":false,"ca_institutions":"Columbia College","funders":"","keywords":"Repurposing; Computer science; Natural language processing; Artificial intelligence; Data science; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01724232,0.001971672,0.001470132,0.01891902,0.003593551,0.01252077,0.005327236,0.003823327,0.02204588],"category_scores_gemma":[0.1464709,0.001315156,0.003341019,0.01225251,0.002741267,0.01677355,0.008878021,0.006716545,0.01614054],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002718399,"about_ca_system_score_gemma":0.009260315,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01394345,"about_ca_topic_score_gemma":0.01349207,"domain_scores_codex":[0.9885685,0.004584359,0.001001558,0.002116481,0.003157431,0.0005715518],"domain_scores_gemma":[0.9197946,0.03707022,0.004350642,0.01617549,0.01936697,0.003242112],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005718213,0.0002677785,0.007483847,0.0021727,0.0003889115,0.001457781,0.005035071,0.007171925,0.005136975,0.05490807,0.3486138,0.5667912],"study_design_scores_gemma":[0.0001445138,0.0002037407,0.004309756,0.001702568,0.0005504883,0.001005463,0.005218021,0.1188183,0.008785657,0.252402,0.6066433,0.0002163585],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04702906,0.02940668,0.6757782,0.1038733,0.03084244,0.001119316,0.0248127,0.02439272,0.06274561],"genre_scores_gemma":[0.3705854,0.0148676,0.4455205,0.01216358,0.01699963,0.001306073,0.05395878,0.01258618,0.07201235],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9827577,"threshold_uncertainty_score":0.09118718,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05142588329128231,"score_gpt":0.3884864920143956,"score_spread":0.3370606087231133,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}