{"id":"W7117578138","doi":"10.1016/j.jss.2025.112763","title":"LeCov: Multi-level testing criteria for large language models","year":2025,"lang":"en","type":"article","venue":"Journal of Systems and Software","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta; Mila - Quebec Artificial Intelligence Institute","funders":"Japan Science and Technology Agency; Japan Society for the Promotion of Science; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Set (abstract data type); Trustworthiness; Prioritization; Test (biology); Test strategy; Risk-based testing; Non-regression testing","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009532462,0.001651682,0.001672558,0.004069523,0.001116632,0.002879313,0.003956275,0.002326296,0.01125695],"category_scores_gemma":[0.07309415,0.0008339647,0.002012988,0.0018579,0.00191481,0.006048703,0.00462452,0.003376033,0.001506378],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002195317,"about_ca_system_score_gemma":0.002797723,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00392352,"about_ca_topic_score_gemma":0.004439214,"domain_scores_codex":[0.9912411,0.003938498,0.0006049761,0.000706153,0.002600663,0.0009085782],"domain_scores_gemma":[0.9284092,0.05580137,0.001826196,0.004100184,0.007887809,0.001975225],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001931955,0.0004375734,0.009438284,0.0009860976,0.0003185569,0.0008069133,0.0003958233,0.3847594,0.01135063,0.2503955,0.0249119,0.3142673],"study_design_scores_gemma":[0.00005483342,0.0001063847,0.000376713,0.00004097269,0.0000266516,0.00006384155,0.00003148434,0.8993604,0.00295486,0.09590202,0.001059223,0.00002258068],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01773428,0.000198896,0.9752514,0.0004148398,0.00005406526,0.0001218511,0.0004505707,0.003471587,0.002302445],"genre_scores_gemma":[0.6342886,0.0001413471,0.3548244,0.0003666867,0.0001459226,0.0005320961,0.002465758,0.00235584,0.004879375],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01125695,"threshold_uncertainty_score":0.05041313,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08902155403723257,"score_gpt":0.3229092410742332,"score_spread":0.2338876870370007,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}