{"id":"W4417462069","doi":"10.2139/ssrn.5735187","title":"Mitigating Lost-in-the-Middle: Dynamic Semantic Chunking for Precise Token-Level Retrieval in RAG","year":2025,"lang":"","type":"preprint","venue":"SSRN Electronic Journal","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Carleton University","funders":"","keywords":"Chunking (psychology); Security token; Coherence (philosophical gambling strategy); Embedding; Merge (version control); Relevance (law); Semantic role labeling","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004062415,0.001058037,0.002023534,0.001875431,0.00145867,0.002984459,0.003285971,0.002252326,0.007594964],"category_scores_gemma":[0.01480929,0.0007798352,0.0008885988,0.002223185,0.001832499,0.009480271,0.006197511,0.002718136,0.004964523],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008815793,"about_ca_system_score_gemma":0.002682025,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004189558,"about_ca_topic_score_gemma":0.005098674,"domain_scores_codex":[0.9974549,0.0008316679,0.0002722723,0.0006122526,0.0004525712,0.0003763271],"domain_scores_gemma":[0.9922691,0.002958146,0.0003026216,0.003487323,0.0006974759,0.0002854873],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001740466,0.0003253616,0.002436013,0.0006096944,0.0001857745,0.0004062562,0.002418761,0.04804781,0.05262877,0.04710237,0.02552164,0.8185771],"study_design_scores_gemma":[0.0001257142,0.0002872818,0.0008324513,0.00008165265,0.0001770246,0.0003091357,0.0008925334,0.7710479,0.05845211,0.1471431,0.02051681,0.0001342244],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02670508,0.001092484,0.9566671,0.0004370383,0.0002721854,0.0001625544,0.0006294981,0.01190616,0.002127885],"genre_scores_gemma":[0.4944763,0.000539417,0.4961806,0.0003882227,0.0002829834,0.0002132576,0.001882794,0.001625394,0.004411032],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007594964,"threshold_uncertainty_score":0.02540773,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03661898335840118,"score_gpt":0.2883529349543292,"score_spread":0.251733951595928,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}