{"id":"W4405095648","doi":"10.48550/arxiv.2412.04403","title":"Establishing Task Scaling Laws via Compute-Efficient Model Ladders","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Distributed and Parallel Computing Systems","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Institute for Catastrophic Loss Reduction","keywords":"Scaling law; Task (project management); Scaling; Computer science; Law; Political science; Mathematics; Economics; Management; Geometry","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0007067692,0.0005712871,0.0005788471,0.0004316031,0.0003131364,0.001096303,0.003478394,0.0004319585,0.000005038301],"category_scores_gemma":[0.00002740519,0.0006531308,0.0004323643,0.001127207,0.0001058703,0.0002392455,0.006414218,0.001322844,0.0002524887],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003533476,"about_ca_system_score_gemma":0.0004186079,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002967674,"about_ca_topic_score_gemma":0.00001543646,"domain_scores_codex":[0.9963353,0.0002017686,0.0004348017,0.002024085,0.0002658367,0.0007381845],"domain_scores_gemma":[0.9973204,0.0001872262,0.0003162077,0.001633256,0.0002217848,0.0003211678],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000005486786,0.00006846934,0.0000410656,0.0001986711,0.00009709764,0.000341511,0.0005872673,0.8784555,0.00001427377,0.117644,0.001997029,0.0005496434],"study_design_scores_gemma":[0.0002396826,0.00001747991,0.00001357719,0.0004339587,0.00006964184,0.0000147077,0.00005016617,0.954904,0.0000116151,0.04291056,0.0006700932,0.0006645581],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07165314,0.00025417,0.918507,0.0001508982,0.002716254,0.000341108,0.00005728552,0.001159396,0.005160707],"genre_scores_gemma":[0.9932492,0.00001668532,0.005344856,0.0001189183,0.0002269649,0.000001142014,0.00006001869,0.00004113321,0.0009410188],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9215961,"threshold_uncertainty_score":0.9999406,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05979223722473317,"score_gpt":0.1873285818208294,"score_spread":0.1275363445960962,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}