{"id":"W4416036956","doi":"10.18653/v1/2025.emnlp-main.285","title":"Tree-of-Quote Prompting Improves Factuality and Attribution in Multi-Hop and Medical Reasoning","year":2025,"lang":"","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Canadian Institutes of Health Research; Deutsche Forschungsgemeinschaft; University of Oxford; Johns Hopkins University","keywords":"Attribution; Natural (archaeology); Natural language; Empirical research; Empirical evidence","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00223212,0.0002245457,0.0004389881,0.0001902206,0.0001496755,0.0001553411,0.000366776,0.0002672139,0.00001962586],"category_scores_gemma":[0.001176858,0.0002111234,0.00004168834,0.0004170678,0.0002086995,0.0004698951,0.001058925,0.0004153686,6.431262e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005864112,"about_ca_system_score_gemma":0.0003188182,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001163305,"about_ca_topic_score_gemma":0.0004990317,"domain_scores_codex":[0.99728,0.0002290252,0.0008374238,0.0008273385,0.000394534,0.0004316503],"domain_scores_gemma":[0.9988501,0.0003171457,0.0001725124,0.0003876479,0.0001022304,0.0001703755],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000135433,0.00008620919,0.04562402,0.0002719023,0.00002020953,0.000007332344,0.002266743,0.00001815671,0.001415925,0.0194054,0.000004264329,0.9308663],"study_design_scores_gemma":[0.00110146,0.00003758715,0.1145157,0.0007103421,0.00001089843,0.000007073314,0.0002751876,0.8808206,0.001790914,0.00052589,0.00003636889,0.0001679893],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4509625,0.0007984168,0.5465398,0.001080494,0.0001579334,0.0002104613,0.000001017647,0.00002761813,0.0002216734],"genre_scores_gemma":[0.9277233,0.0001580757,0.07168719,0.0001437424,0.00002367956,0.000006829649,9.070593e-7,0.000004502988,0.0002517633],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9306983,"threshold_uncertainty_score":0.8609364,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03687982815593502,"score_gpt":0.3170794861581848,"score_spread":0.2801996580022498,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}