{"id":"W4416746178","doi":"10.1016/j.parco.2025.103165","title":"Butterfly factorization for vision transformers on multi-IPU systems","year":2025,"lang":"en","type":"article","venue":"Parallel Computing","topic":"Advanced Memory and Neural Computing","field":"Engineering","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Butterfly; Memory footprint; Transformer; Factorization; Computation","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002945713,0.0005826548,0.0004080151,0.0003865843,0.0003493901,0.0007315825,0.0009219017,0.0005371412,0.006463278],"category_scores_gemma":[0.001554693,0.0002713607,0.0004601412,0.0004631803,0.0004232805,0.001459762,0.0006346143,0.0008332914,0.00120307],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009646837,"about_ca_system_score_gemma":0.0009903399,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00778245,"about_ca_topic_score_gemma":0.01135436,"domain_scores_codex":[0.9997928,0.00003514542,0.00001164565,0.00004456946,0.00007863839,0.00003725335],"domain_scores_gemma":[0.9996728,0.0001195993,0.00002465934,0.00008347497,0.00007454435,0.00002491949],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005581754,0.0001265599,0.003063046,0.00027163,0.00007392391,0.0005687376,0.0002691596,0.5436905,0.03640322,0.05091356,0.01630553,0.3477559],"study_design_scores_gemma":[0.00001048961,0.00004004898,0.0002036621,0.000006958632,0.000004857677,0.00004338019,0.00003180671,0.9857449,0.005847408,0.005690284,0.002370964,0.000005303526],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1433683,0.0005923475,0.8362958,0.0005381671,0.0001387478,0.00009624602,0.0003675341,0.005931061,0.01267168],"genre_scores_gemma":[0.6857207,0.0002611295,0.3076859,0.0002064602,0.00003000981,0.00009766365,0.0005547389,0.000277808,0.005165644],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00778245,"threshold_uncertainty_score":0.02162188,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02213822854525998,"score_gpt":0.2953664193748585,"score_spread":0.2732281908295985,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}