{"id":"W4415524320","doi":"10.1109/mlsp62443.2025.11204231","title":"Colflor: Towards Bert-Size Vision-Language Document Retrieval Models","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"","keywords":"Document retrieval; Encoding (memory); Process (computing); Image retrieval; Document clustering; Visual Word; Infographic; Data retrieval; Vector space model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002167347,0.001586731,0.001885193,0.00259441,0.0006510293,0.002767018,0.004399075,0.002535957,0.006189877],"category_scores_gemma":[0.008690778,0.0008057331,0.002023118,0.002135416,0.0006961869,0.004814075,0.001611897,0.002077132,0.006839546],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002365964,"about_ca_system_score_gemma":0.001739003,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0166014,"about_ca_topic_score_gemma":0.01684789,"domain_scores_codex":[0.9986808,0.0002665651,0.0001029082,0.0003999781,0.0004314331,0.0001183872],"domain_scores_gemma":[0.9971982,0.001249415,0.0001983541,0.000526734,0.0007291219,0.00009801718],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009264685,0.0005277988,0.00147898,0.0007430571,0.0002746843,0.0002644646,0.0001948751,0.2209564,0.01953228,0.01723316,0.05948531,0.6783826],"study_design_scores_gemma":[0.0000713744,0.000121373,0.0002652764,0.00003035183,0.00004081996,0.0001735262,0.00003077719,0.9755568,0.005051267,0.009795401,0.008828717,0.00003430455],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01674702,0.002670835,0.9542863,0.0009416185,0.0002049042,0.0005338166,0.002549407,0.0179077,0.004158431],"genre_scores_gemma":[0.16038,0.002308539,0.8086334,0.001522382,0.0003693646,0.001107289,0.008713573,0.001471585,0.01549376],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0166014,"threshold_uncertainty_score":0.03300959,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.009472087359197887,"score_gpt":0.309669048924067,"score_spread":0.3001969615648691,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}