{"id":"W4405034991","doi":"10.48550/arxiv.2412.01690","title":"Can We Afford The Perfect Prompt? Balancing Cost and Accuracy with the Economical Prompting Index","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Economic, financial, and policy analysis","field":"Economics, Econometrics and Finance","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Index (typography); Computer science; Economics; Programming language","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0007434714,0.0004387822,0.000709704,0.0003027905,0.0004786765,0.0004110501,0.0008241993,0.0002830625,0.0001087559],"category_scores_gemma":[0.00006434033,0.0003482075,0.0002975877,0.0003386475,0.0003292754,0.0001626508,0.001030025,0.001168019,0.0002040312],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003845568,"about_ca_system_score_gemma":0.0001928117,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002634972,"about_ca_topic_score_gemma":0.003423987,"domain_scores_codex":[0.9977593,0.00006041175,0.0004533211,0.001187324,0.00001935443,0.0005202726],"domain_scores_gemma":[0.9979854,0.0003245872,0.0006391172,0.0008853583,0.00003345893,0.000132142],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001914241,0.0000854601,0.1783271,0.0005708819,0.001500064,0.00009515813,0.003934394,0.07817473,0.000004337493,0.7334792,0.001824213,0.001812996],"study_design_scores_gemma":[0.001623082,0.0002117934,0.03600892,0.000474073,0.0007406642,0.00004623521,0.002007884,0.5396544,0.00004238833,0.3636506,0.05318401,0.002355992],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9834759,0.001047112,0.0008088929,0.005602038,0.0002916018,0.0007935152,0.0002847044,0.0000735853,0.00762266],"genre_scores_gemma":[0.9960495,0.001021971,0.00002048752,0.0002900102,0.0003569429,0.00001590932,0.00002106961,0.00005529574,0.002168835],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4614796,"threshold_uncertainty_score":0.999897,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06088055200143749,"score_gpt":0.1812284313605751,"score_spread":0.1203478793591377,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}