{"id":"W4416677522","doi":"10.1109/models67397.2025.00024","title":"SHERPA: A Model-Driven Framework for Large Language Model Execution","year":2025,"lang":"","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Variety (cybernetics); State (computer science); Class (philosophy); Structuring; Best practice; Baseline (sea); Language model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004209899,0.002220729,0.001230492,0.001480601,0.0009234806,0.002818485,0.004835788,0.001857244,0.008848342],"category_scores_gemma":[0.01504654,0.001868996,0.00327005,0.0009887528,0.001272097,0.00316678,0.003414187,0.005026918,0.004436016],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001535172,"about_ca_system_score_gemma":0.004824804,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01112499,"about_ca_topic_score_gemma":0.0247368,"domain_scores_codex":[0.9975968,0.001049761,0.0001857232,0.0004829432,0.0005593974,0.0001253333],"domain_scores_gemma":[0.9936354,0.004460439,0.000290381,0.0008225817,0.0005748472,0.0002163655],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004730589,0.0002855679,0.002399218,0.0008560572,0.0004130178,0.0004088274,0.0006057366,0.6872778,0.007122461,0.07720851,0.03374,0.1892098],"study_design_scores_gemma":[0.00002859888,0.00001730701,0.00003741297,0.00001495345,0.00001369789,0.00001832738,0.00001228133,0.972329,0.001147925,0.02121806,0.005148024,0.00001437367],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001126213,0.0001440914,0.9692402,0.0001610368,0.00004977396,0.000145759,0.0005124381,0.02801206,0.0006084937],"genre_scores_gemma":[0.06283162,0.0003399174,0.9264166,0.0003052444,0.00006748416,0.0008991231,0.003224696,0.004392088,0.001523273],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01112499,"threshold_uncertainty_score":0.02960062,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02454095306884229,"score_gpt":0.3090076620587478,"score_spread":0.2844667089899055,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}