{"id":"W4401408800","doi":"10.1145/3673038.3673124","title":"Arlo: Serving Transformer-based Language Models with Dynamic Input Lengths","year":2024,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Microsoft (Canada)","funders":"","keywords":"Computer science; Latency (audio); Compiler; Padding; Scheduling (production processes); Parallel computing; Testbed; Distributed computing; Queue; Runtime system; Serialization; Programming language; Computer network; Mathematical optimization","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001511508,0.0001349352,0.0001118387,0.0001153087,0.00005184272,0.0002469447,0.0005255205,0.00004540617,0.00003750164],"category_scores_gemma":[0.000002126532,0.00009926148,0.00004916975,0.0003026554,0.00001423244,0.0007122501,0.00003465573,0.0001553754,0.0000398927],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005253691,"about_ca_system_score_gemma":0.0001343894,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001407387,"about_ca_topic_score_gemma":0.0002069163,"domain_scores_codex":[0.9989209,0.00001854923,0.0001477415,0.0004050976,0.0002458929,0.0002618677],"domain_scores_gemma":[0.9993894,0.00005280003,0.00001192121,0.0004558123,0.00002247195,0.00006765183],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001125766,0.00006611511,0.00003616898,0.0004140987,0.00006653344,0.0003194524,0.009245476,0.2845716,0.004495739,0.2487155,0.0001288307,0.4519292],"study_design_scores_gemma":[0.0001465201,0.00003442994,0.00000649762,0.00009347383,0.000005788338,0.00001175511,0.0001053453,0.996386,0.001159923,0.001725601,0.0001661967,0.0001585011],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04794471,0.0003399285,0.9379241,0.001808722,0.0001396762,0.0001179529,0.000001390384,0.0007411457,0.01098241],"genre_scores_gemma":[0.8593677,0.000004309858,0.1392356,0.0004292953,0.00002068873,0.00001084113,0.000001757114,0.00001402611,0.0009158385],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8114229,"threshold_uncertainty_score":0.4047766,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01136758055846535,"score_gpt":0.2410187164644272,"score_spread":0.2296511359059619,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}