{"id":"W4417469644","doi":"10.48550/arxiv.2512.14427","title":"Effect of Document Packing on the Latent Multi-Hop Reasoning Capabilities of Large Language Models","year":2025,"lang":"","type":"preprint","venue":"ArXiv.org","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Key (lock); Process (computing); Language model; Language understanding; Data modeling; Training set","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01043419,0.001168031,0.001262231,0.000911964,0.001321171,0.002481846,0.001395672,0.002018164,0.00269599],"category_scores_gemma":[0.1115291,0.000989815,0.0008301337,0.0009885578,0.002210099,0.006896983,0.00294066,0.003768166,0.0006341182],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001233824,"about_ca_system_score_gemma":0.001747497,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004344225,"about_ca_topic_score_gemma":0.004393838,"domain_scores_codex":[0.9962274,0.002126444,0.0002863895,0.0005614468,0.0003751779,0.0004231392],"domain_scores_gemma":[0.8435153,0.1407098,0.003075183,0.008629456,0.002047983,0.0020223],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005315095,0.001237525,0.02736812,0.0007242005,0.0003643421,0.0005326353,0.002016704,0.70223,0.01761037,0.01340919,0.005565889,0.223626],"study_design_scores_gemma":[0.0001768382,0.0007018236,0.003874552,0.0001018144,0.0001412082,0.0001882229,0.0005093213,0.9629323,0.01276952,0.01751408,0.001037813,0.00005252008],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9274619,0.001456967,0.06379548,0.001434295,0.0001276687,0.0001012954,0.000299655,0.001384484,0.003938218],"genre_scores_gemma":[0.9743251,0.0003517634,0.02378609,0.0002276212,0.00004846932,0.00006975915,0.0003777457,0.000205509,0.0006080415],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01043419,"threshold_uncertainty_score":0.05518198,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03533297112411127,"score_gpt":0.3004903172743915,"score_spread":0.2651573461502802,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}