{"id":"W4401408800","doi":"10.1145/3673038.3673124","title":"Arlo: Serving Transformer-based Language Models with Dynamic Input Lengths","year":2024,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Microsoft (Canada)","funders":"","keywords":"Computer science; Latency (audio); Compiler; Padding; Scheduling (production processes); Parallel computing; Testbed; Distributed computing; Queue; Runtime system; Serialization; Programming language; Computer network; Mathematical optimization","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001463739,0.0009484487,0.0006697579,0.0006090303,0.0004728977,0.001611708,0.002688143,0.0007634346,0.004592985],"category_scores_gemma":[0.005844004,0.0007829973,0.001097598,0.0007120428,0.0008046267,0.00347189,0.001762169,0.00155343,0.002290414],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001327142,"about_ca_system_score_gemma":0.002634641,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008936595,"about_ca_topic_score_gemma":0.02252928,"domain_scores_codex":[0.9990355,0.0002541948,0.00007435281,0.0002602561,0.0002539227,0.0001218135],"domain_scores_gemma":[0.996758,0.001535167,0.0001793516,0.001082059,0.0003057919,0.000139622],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002188067,0.001015804,0.01083465,0.001011863,0.0003208855,0.0007653072,0.001502811,0.4804338,0.1020452,0.05130155,0.05282442,0.2957556],"study_design_scores_gemma":[0.00006348155,0.00005955395,0.0001713235,0.000005679061,0.00002010946,0.00004827561,0.00006994568,0.9766631,0.01183835,0.006350347,0.004691045,0.00001872385],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.05947436,0.0002163398,0.8076403,0.0004942489,0.0001066921,0.000252005,0.001709828,0.1250736,0.005032703],"genre_scores_gemma":[0.562162,0.0002614688,0.4204668,0.0004441098,0.00006955232,0.0002831146,0.004346842,0.006407539,0.005558572],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008936595,"threshold_uncertainty_score":0.01776916,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01136758055846535,"score_gpt":0.2410187164644272,"score_spread":0.2296511359059619,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}