{"id":"W7117578138","doi":"10.1016/j.jss.2025.112763","title":"LeCov: Multi-level testing criteria for large language models","year":2025,"lang":"en","type":"article","venue":"Journal of Systems and Software","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta; Mila - Quebec Artificial Intelligence Institute","funders":"Japan Science and Technology Agency; Japan Society for the Promotion of Science; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Set (abstract data type); Trustworthiness; Prioritization; Test (biology); Test strategy; Risk-based testing; Non-regression testing","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000633068,0.000078936,0.0002175826,0.0001065032,0.0001081231,0.0001858863,0.0002838343,0.00005036246,4.439869e-7],"category_scores_gemma":[0.0002740083,0.00006332195,0.00005222577,0.00009703779,0.000004860337,0.0003863674,0.00008758715,0.00009304658,2.63505e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002687858,"about_ca_system_score_gemma":0.00008370657,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003076647,"about_ca_topic_score_gemma":0.00000450927,"domain_scores_codex":[0.9991955,0.00003129163,0.0003689184,0.0001283771,0.000114934,0.0001609429],"domain_scores_gemma":[0.9991537,0.0001847033,0.0001839246,0.0001640229,0.0002592676,0.0000544377],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001231172,0.0008021591,0.0381078,0.008296835,0.0008016385,0.0006068425,0.03857403,0.04732503,0.01114138,0.2096257,0.01568366,0.6289118],"study_design_scores_gemma":[0.00103364,0.00005704943,0.0008569997,0.0007013039,0.00001376722,0.0001075098,0.0004456618,0.9941728,0.00004726853,0.001568468,0.0008828529,0.0001126656],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02425302,0.002719983,0.9719909,0.0001116263,0.0007363906,0.0001129468,0.00001065732,0.00002919255,0.00003532297],"genre_scores_gemma":[0.625084,0.000006064005,0.3742541,0.00009973527,0.0001198268,0.000004023015,2.490404e-7,0.000004588173,0.0004274157],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9468478,"threshold_uncertainty_score":0.2582194,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08902155403723257,"score_gpt":0.3229092410742332,"score_spread":0.2338876870370007,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}