{"id":"W4405095648","doi":"10.48550/arxiv.2412.04403","title":"Establishing Task Scaling Laws via Compute-Efficient Model Ladders","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Distributed and Parallel Computing Systems","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Institute for Catastrophic Loss Reduction","keywords":"Scaling law; Task (project management); Scaling; Computer science; Law; Political science; Mathematics; Economics; Management; Geometry","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004390752,0.001364326,0.0014129,0.001106412,0.0006643941,0.001727754,0.002617928,0.001226718,0.004488005],"category_scores_gemma":[0.03051449,0.001148107,0.001365741,0.0008662005,0.001544417,0.004652358,0.002910994,0.003735857,0.001464279],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002120073,"about_ca_system_score_gemma":0.002954801,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009650837,"about_ca_topic_score_gemma":0.01085994,"domain_scores_codex":[0.9980932,0.0004933416,0.0001182372,0.0004668259,0.0005518707,0.0002765148],"domain_scores_gemma":[0.9864712,0.008067638,0.001090423,0.002523953,0.001370052,0.0004766781],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001488365,0.000138818,0.002653957,0.0000703865,0.00003307981,0.00005640239,0.0001366197,0.9263548,0.001991073,0.01598565,0.0019121,0.05051823],"study_design_scores_gemma":[0.000005476395,0.00001109329,0.0000820903,0.000003710971,0.000002515422,0.000003043138,0.000003962733,0.9917471,0.0002599082,0.007770638,0.0001072095,0.000003089917],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04059743,0.0001421183,0.9545103,0.0003839983,0.00003653918,0.000116687,0.0002350179,0.002031264,0.001946617],"genre_scores_gemma":[0.7052037,0.0002636358,0.2877058,0.0003855039,0.00007116976,0.0008977464,0.0008683243,0.0008619968,0.003742144],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009650837,"threshold_uncertainty_score":0.02322078,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05979223722473317,"score_gpt":0.1873285818208294,"score_spread":0.1275363445960962,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}