{"id":"W4413002442","doi":"10.3389/frai.2025.1592013","title":"Large language models for closed-library multi-document query, test generation, and evaluation","year":2025,"lang":"en","type":"article","venue":"Frontiers in Artificial Intelligence","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Massachusetts Institute of Technology","keywords":"Computer science; Knowledge base; Leverage (statistics); Set (abstract data type); Information retrieval; World Wide Web; Language model; Data science; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01515414,0.001953662,0.001701937,0.00235648,0.0007875046,0.003350881,0.003948725,0.003371797,0.01296897],"category_scores_gemma":[0.04989192,0.0007989362,0.002153884,0.002131887,0.001262724,0.003629133,0.002895717,0.003045568,0.006455894],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004059239,"about_ca_system_score_gemma":0.002929249,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01186312,"about_ca_topic_score_gemma":0.01096521,"domain_scores_codex":[0.9869938,0.007724641,0.0009390887,0.001629692,0.00225482,0.0004579508],"domain_scores_gemma":[0.9630075,0.02763511,0.001346307,0.003372588,0.003958421,0.0006801683],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001491848,0.0009247357,0.003645442,0.001158309,0.0003369951,0.0003355427,0.0003624841,0.4165642,0.003294696,0.02592263,0.0489208,0.4970423],"study_design_scores_gemma":[0.00008966911,0.0001163608,0.000396049,0.00005252792,0.00002960362,0.00007919363,0.00004292588,0.9759322,0.001362345,0.01822986,0.003643865,0.00002543066],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02105278,0.002466014,0.9462829,0.001294939,0.0002237139,0.001411319,0.004442095,0.01772259,0.005103666],"genre_scores_gemma":[0.3579359,0.0007049435,0.6134058,0.0009590129,0.0002896641,0.003387264,0.0158579,0.001473169,0.005986351],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01515414,"threshold_uncertainty_score":0.08014369,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05380820147992983,"score_gpt":0.3267508589186304,"score_spread":0.2729426574387006,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}