{"id":"W4414004064","doi":"10.14778/3749646.3749702","title":"ThriftLLM: On Cost-Effective Selection of Large Language Models for Classification Queries","year":2025,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Selection (genetic algorithm); Computer science; Natural language processing; Artificial intelligence; Information retrieval","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006531773,0.003016382,0.003507783,0.001912142,0.001616673,0.003047073,0.004863843,0.003064694,0.004626693],"category_scores_gemma":[0.02322312,0.00109811,0.002760867,0.002807314,0.00123189,0.005387856,0.005029688,0.004279564,0.001946508],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003615819,"about_ca_system_score_gemma":0.004265406,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01257133,"about_ca_topic_score_gemma":0.01689307,"domain_scores_codex":[0.9942926,0.00186816,0.0003638317,0.001086648,0.001583475,0.0008052532],"domain_scores_gemma":[0.9835699,0.01219803,0.0007361291,0.001880932,0.001042537,0.0005724457],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009018473,0.0008895669,0.006349277,0.0003784643,0.0002522273,0.000456096,0.0005207497,0.5439692,0.00444036,0.01870561,0.03425291,0.3888837],"study_design_scores_gemma":[0.00004192943,0.00004926793,0.0001199707,0.000006909867,0.00001790088,0.00005511467,0.00004888979,0.9897236,0.0004979025,0.008776002,0.0006543993,0.000008099775],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.06150642,0.001598215,0.9186172,0.002612154,0.0001478318,0.0006784962,0.001019681,0.0104021,0.00341797],"genre_scores_gemma":[0.4729928,0.0005788163,0.5124458,0.002008243,0.0003787718,0.0008683321,0.004803001,0.001283269,0.004640845],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01257133,"threshold_uncertainty_score":0.03454375,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09834815259862227,"score_gpt":0.4145340017749151,"score_spread":0.3161858491762928,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}