{"id":"W4413002442","doi":"10.3389/frai.2025.1592013","title":"Large language models for closed-library multi-document query, test generation, and evaluation","year":2025,"lang":"en","type":"article","venue":"Frontiers in Artificial Intelligence","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Massachusetts Institute of Technology","keywords":"Computer science; Knowledge base; Leverage (statistics); Set (abstract data type); Information retrieval; World Wide Web; Language model; Data science; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000709655,0.0001219196,0.0001493111,0.0002158109,0.0001233467,0.0002054014,0.0004000891,0.00007473014,0.000006871737],"category_scores_gemma":[0.0001737934,0.0001290216,0.00003554853,0.000308561,0.00003153624,0.0007909195,0.0001546535,0.00009655415,0.000002608519],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007434412,"about_ca_system_score_gemma":0.000132279,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002840801,"about_ca_topic_score_gemma":0.0000978624,"domain_scores_codex":[0.998632,0.00007614726,0.0003888425,0.0004680144,0.0001809964,0.0002539802],"domain_scores_gemma":[0.9993827,0.00008504664,0.00006038942,0.000357278,0.000071451,0.00004317919],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000128839,0.0001611962,0.00145121,0.00003762181,0.00001520503,0.000002995713,0.004568938,0.04626478,0.00113667,0.2847785,0.003144482,0.6584255],"study_design_scores_gemma":[0.00007225502,0.0000164841,0.00003590624,0.0000248097,0.000005991196,2.911635e-7,0.0004675079,0.8422197,0.01625004,0.1405533,0.0002475278,0.0001061282],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01351075,0.001417075,0.9818733,0.00103444,0.001194106,0.000669913,0.000007989816,0.00007238275,0.0002200759],"genre_scores_gemma":[0.6337592,0.00005153795,0.3653668,0.0003444135,0.00007586105,0.0001187632,0.00001257261,0.000006657726,0.0002642245],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.7959549,"threshold_uncertainty_score":0.5261348,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05380820147992983,"score_gpt":0.3267508589186304,"score_spread":0.2729426574387006,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}