{"id":"W4412944489","doi":"10.18653/v1/2025.findings-acl.1312","title":"MDBench: A Synthetic Multi-Document Reasoning Benchmark Generated with Knowledge Guidance","year":2025,"lang":"en","type":"article","venue":"","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Institute for Catastrophic Loss Reduction; Cisco Systems","keywords":"Benchmark (surveying); Computer science; Artificial intelligence; Model-based reasoning; Information retrieval; Knowledge representation and reasoning","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003999408,0.001476723,0.0005496984,0.002185612,0.001056646,0.002115004,0.003027726,0.002259532,0.005492803],"category_scores_gemma":[0.02544319,0.0003550828,0.001363414,0.002034175,0.001095249,0.00267507,0.001975769,0.002413317,0.002263057],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001942567,"about_ca_system_score_gemma":0.002177438,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01020559,"about_ca_topic_score_gemma":0.01918465,"domain_scores_codex":[0.9953189,0.002016307,0.0004283444,0.0009189791,0.001140394,0.000177047],"domain_scores_gemma":[0.9806125,0.01299816,0.0005447952,0.002977265,0.002346081,0.0005213273],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001778418,0.003098316,0.01317883,0.006144494,0.0006041997,0.001878159,0.001552969,0.1873863,0.02151026,0.03196633,0.3698815,0.3610202],"study_design_scores_gemma":[0.001031516,0.001217528,0.007447057,0.0005357145,0.0001821997,0.001107037,0.001446262,0.6832479,0.05696761,0.04470619,0.2019176,0.0001933912],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"empirical","genre_gemma":"dataset","genre_scores_codex":[0.3877844,0.007319625,0.2959614,0.0063843,0.002199524,0.003528534,0.1581971,0.07221272,0.06641239],"genre_scores_gemma":[0.3490218,0.001031006,0.3867182,0.001515109,0.000189524,0.00137549,0.248537,0.002611968,0.008999975],"genre_candidate":"dataset","genre_consensus":null,"teacher_disagreement_score":0.01020559,"threshold_uncertainty_score":0.02115119,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01474922071863538,"score_gpt":0.2714078627290791,"score_spread":0.2566586420104437,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}