{"id":"W4405034991","doi":"10.48550/arxiv.2412.01690","title":"Can We Afford The Perfect Prompt? Balancing Cost and Accuracy with the Economical Prompting Index","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Economic, financial, and policy analysis","field":"Economics, Econometrics and Finance","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Index (typography); Computer science; Economics; Programming language","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01853153,0.001331984,0.00146669,0.002553321,0.0009328155,0.005841949,0.00159687,0.002128083,0.005000975],"category_scores_gemma":[0.1653395,0.0005273065,0.0007008624,0.003071786,0.001797234,0.01292375,0.002688343,0.003051185,0.001990869],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001619778,"about_ca_system_score_gemma":0.002797535,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002812126,"about_ca_topic_score_gemma":0.003054844,"domain_scores_codex":[0.9869773,0.007890736,0.0009558759,0.00133882,0.002280509,0.0005567833],"domain_scores_gemma":[0.8753583,0.1000518,0.004551922,0.01223368,0.00578403,0.002020239],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.006192973,0.0007080419,0.0845165,0.00162673,0.0003721524,0.0003419106,0.001824965,0.1311074,0.00683582,0.04758594,0.03176811,0.6871194],"study_design_scores_gemma":[0.0006100807,0.001571555,0.02378831,0.0004622431,0.0003954914,0.0006599761,0.002867215,0.665128,0.01879815,0.2509615,0.03440983,0.0003476403],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4297084,0.005860987,0.5139135,0.01290293,0.0008034612,0.0004830155,0.005101996,0.01325183,0.01797389],"genre_scores_gemma":[0.8602983,0.0008246604,0.1336168,0.0005260081,0.0001987475,0.0001755974,0.001986747,0.0006590152,0.001714111],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01853153,"threshold_uncertainty_score":0.09800529,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06088055200143749,"score_gpt":0.1812284313605751,"score_spread":0.1203478793591377,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}