{"id":"W4416036956","doi":"10.18653/v1/2025.emnlp-main.285","title":"Tree-of-Quote Prompting Improves Factuality and Attribution in Multi-Hop and Medical Reasoning","year":2025,"lang":"","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Canadian Institutes of Health Research; Deutsche Forschungsgemeinschaft; University of Oxford; Johns Hopkins University","keywords":"Attribution; Natural (archaeology); Natural language; Empirical research; Empirical evidence","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007648317,0.0007965933,0.0009993379,0.001880234,0.001041969,0.002155256,0.001648284,0.00190761,0.01099647],"category_scores_gemma":[0.05488503,0.0004864119,0.0008363994,0.001317641,0.0006321233,0.008304112,0.003428452,0.002422376,0.002372169],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008105796,"about_ca_system_score_gemma":0.002166119,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003536375,"about_ca_topic_score_gemma":0.005981509,"domain_scores_codex":[0.9958873,0.001977615,0.0003691326,0.0009213815,0.000675322,0.0001692667],"domain_scores_gemma":[0.9582378,0.03197791,0.002035452,0.004216541,0.002383114,0.001149206],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002837862,0.001173145,0.01645299,0.0007058345,0.0001801396,0.0003187554,0.001519891,0.01913476,0.005748993,0.0125716,0.03002147,0.9093346],"study_design_scores_gemma":[0.0008068041,0.001022832,0.01544341,0.0004708363,0.0006954062,0.0005982231,0.001847265,0.7205343,0.02817567,0.2004565,0.02974632,0.0002024746],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3066036,0.005761757,0.6271674,0.008437752,0.001868964,0.0007208913,0.003411073,0.02936761,0.01666101],"genre_scores_gemma":[0.7822768,0.0007716443,0.209517,0.0006063167,0.0002626582,0.00007439002,0.002835713,0.0004759361,0.003179349],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01099647,"threshold_uncertainty_score":0.04044867,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03687982815593502,"score_gpt":0.3170794861581848,"score_spread":0.2801996580022498,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}