{"id":"W4417462069","doi":"10.2139/ssrn.5735187","title":"Mitigating Lost-in-the-Middle: Dynamic Semantic Chunking for Precise Token-Level Retrieval in RAG","year":2025,"lang":"","type":"preprint","venue":"SSRN Electronic Journal","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Carleton University","funders":"","keywords":"Chunking (psychology); Security token; Coherence (philosophical gambling strategy); Embedding; Merge (version control); Relevance (law); Semantic role labeling","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication","open_science","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.02035672,0.001091142,0.001401242,0.001298909,0.0007575976,0.001183405,0.006560331,0.0007993396,0.00001549437],"category_scores_gemma":[0.001273192,0.001061357,0.0008034949,0.00170004,0.000133595,0.0008002145,0.001613936,0.01578641,0.00001559246],"about_ca_system_candidate":true,"about_ca_system_consensus":true,"about_ca_system_score_codex":0.007352873,"about_ca_system_score_gemma":0.01838611,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002213491,"about_ca_topic_score_gemma":0.004979549,"domain_scores_codex":[0.9842416,0.001151083,0.003105459,0.002070523,0.001532033,0.007899292],"domain_scores_gemma":[0.9946018,0.001429187,0.00154789,0.001768658,0.0004616251,0.000190836],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000918674,0.001061659,0.00311083,0.002424558,0.0009560373,0.0002824613,0.03162883,0.1609537,0.0007335846,0.2426129,0.00003016927,0.5552865],"study_design_scores_gemma":[0.002354276,0.0002560864,0.0004434646,0.003584755,0.0001054969,0.0007994894,0.002054873,0.6241775,0.00007607887,0.3652703,0.00007977278,0.0007979423],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2155027,0.008114233,0.7673978,0.003639731,0.00265495,0.002199343,0.00001939788,0.00004002886,0.0004318765],"genre_scores_gemma":[0.9759391,0.006613446,0.01382939,0.0003299977,0.000727062,0.00007233776,0.00001398614,0.00006674899,0.002407921],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7604364,"threshold_uncertainty_score":0.9998535,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03661898335840118,"score_gpt":0.2883529349543292,"score_spread":0.251733951595928,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}