{"id":"W4416677522","doi":"10.1109/models67397.2025.00024","title":"SHERPA: A Model-Driven Framework for Large Language Model Execution","year":2025,"lang":"","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Variety (cybernetics); State (computer science); Class (philosophy); Structuring; Best practice; Baseline (sea); Language model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0005068998,0.000367352,0.0004199488,0.0002177739,0.000395861,0.0003670439,0.001372577,0.0004672893,0.00004336016],"category_scores_gemma":[0.0001565868,0.0003756825,0.0002905943,0.0005133267,0.00003309164,0.0006245687,0.0008153513,0.0004428762,0.00002843695],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001978046,"about_ca_system_score_gemma":0.0005366739,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003709885,"about_ca_topic_score_gemma":0.00004229258,"domain_scores_codex":[0.996911,0.00005459589,0.0006227249,0.001143141,0.0003601941,0.0009083771],"domain_scores_gemma":[0.9978672,0.0001639024,0.000129741,0.001469434,0.0002153596,0.0001543174],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001436002,0.00009458228,0.00001192909,0.00007316981,0.00002759856,0.000001384776,0.002779824,0.3837698,0.0001423682,0.5994508,0.0008508004,0.01278336],"study_design_scores_gemma":[0.000571239,0.00002763275,0.000002525974,0.0002033461,0.0000405053,6.963909e-7,0.0001903571,0.7910272,0.0003590631,0.2070569,0.0002286622,0.0002918897],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003217577,0.0006365266,0.9826655,0.003607023,0.0008406007,0.0009277752,0.00004050024,0.0002918016,0.00777272],"genre_scores_gemma":[0.4726804,0.00002623573,0.515746,0.002006585,0.00008647278,0.00007633004,0.000003972415,0.00001376285,0.009360196],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.4694628,"threshold_uncertainty_score":0.9998695,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02454095306884229,"score_gpt":0.3090076620587478,"score_spread":0.2844667089899055,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}