{"id":"W6979310868","doi":"","title":"The Great Nugget Recall: Automating Fact Extraction and RAG Evaluation with Large Language Models","year":2025,"lang":"en","type":"article","venue":"ArXiv.org","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Institute for Information and Communications Technology Promotion; Natural Sciences and Engineering Research Council of Canada; National Institute of Standards and Technology; Ministry of Science and ICT, South Korea","keywords":"Context (archaeology); Focus (optics); Quality (philosophy); Information extraction; Work (physics); Language model; Question answering; Track (disk drive)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03939525,0.00213751,0.001455224,0.00455383,0.001451598,0.005792429,0.003264736,0.002447538,0.002888561],"category_scores_gemma":[0.09198347,0.001305164,0.001903854,0.002025513,0.001800509,0.008818558,0.005424964,0.003628894,0.001933166],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0027246,"about_ca_system_score_gemma":0.003486675,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01019271,"about_ca_topic_score_gemma":0.0174178,"domain_scores_codex":[0.9545275,0.02997223,0.0024195,0.005291311,0.00689865,0.000890835],"domain_scores_gemma":[0.9003869,0.06653742,0.004995762,0.01861259,0.008226438,0.001240979],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001278511,0.0008487383,0.03652295,0.001777524,0.00125926,0.0006074788,0.006877974,0.08605324,0.05097518,0.0206651,0.03191794,0.7612161],"study_design_scores_gemma":[0.0002061345,0.0007394284,0.01552278,0.0002910022,0.000381091,0.0005040161,0.001074007,0.859409,0.06736725,0.02450194,0.02968282,0.0003205538],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1151555,0.001449356,0.8284335,0.001251439,0.0001907292,0.0009262387,0.002178384,0.04519376,0.005221106],"genre_scores_gemma":[0.4701355,0.0002178199,0.5186014,0.0004893755,0.00008065152,0.0006255964,0.005283726,0.002711448,0.001854408],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.03939525,"threshold_uncertainty_score":0.2083445,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03968283115839261,"score_gpt":0.3117375363597709,"score_spread":0.2720547052013783,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}