{"id":"W4406858164","doi":"10.1109/tse.2025.3533972","title":"On the Workflows and Smells of Leaderboard Operations (LBOps): An Exploratory Study of Foundation Model Leaderboards","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Human Resource Development and Performance Evaluation","field":"Psychology","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Software engineering; Workflow; Programming language; Foundation (evidence); World Wide Web; Database","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.006122653,0.0004057395,0.0002530541,0.002386744,0.001737694,0.002914248,0.0009243177,0.0006371678,0.001351582],"category_scores_gemma":[0.02726777,0.0003155136,0.0003489364,0.002397136,0.002023258,0.003532137,0.002540761,0.001004906,0.0004549042],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001841164,"about_ca_system_score_gemma":0.002654404,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00680108,"about_ca_topic_score_gemma":0.009972148,"domain_scores_codex":[0.9960944,0.002026487,0.0002288913,0.0004992817,0.0008079979,0.000342883],"domain_scores_gemma":[0.9680179,0.0206809,0.004455388,0.002739713,0.002870631,0.001235556],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0004587919,0.0009121645,0.3359072,0.0008796151,0.00006156079,0.003635608,0.3994184,0.007382766,0.01336116,0.01996669,0.00797281,0.2100431],"study_design_scores_gemma":[0.00007312002,0.000704563,0.325559,0.001058105,0.00006700418,0.001558361,0.4937123,0.05955701,0.01540231,0.02600201,0.07599115,0.000315109],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9819411,0.00008921952,0.01386688,0.0002471893,0.00000729047,0.0001571503,0.0002985484,0.0001353593,0.003257201],"genre_scores_gemma":[0.973918,0.0001389152,0.02300976,0.00009332942,0.000007073631,0.0001958009,0.0008538021,0.0001340646,0.001649194],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9938774,"threshold_uncertainty_score":0.03238004,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04758252356410681,"score_gpt":0.3070713554868336,"score_spread":0.2594888319227268,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}