{"id":"W4213397781","doi":"","title":"Mining Test Repositories for Automatic Detection of UI Performance Regressions in Android Apps","year":2016,"lang":"en","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Android (operating system); Computer science; Test (biology); Mobile apps; World Wide Web; Operating system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003091652,0.0001190279,0.0001962779,0.0001357676,0.0002289673,0.00006674382,0.0006622064,0.00008572311,0.000003763265],"category_scores_gemma":[0.002076351,0.00008669824,0.00006892217,0.0004814913,0.0001146593,0.0004786182,0.0001943163,0.00006960093,0.000004849859],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007449748,"about_ca_system_score_gemma":0.0001034786,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007207145,"about_ca_topic_score_gemma":0.0001921669,"domain_scores_codex":[0.9981601,0.000601771,0.0004603058,0.0003440862,0.0002163492,0.0002174313],"domain_scores_gemma":[0.9953895,0.002374881,0.0003058124,0.001092233,0.0007761539,0.00006139473],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00001238709,0.0003906956,0.1739317,0.0003640007,0.00001537267,0.000001045321,0.00864394,0.000009616331,0.07958037,0.003771028,0.0001243289,0.7331555],"study_design_scores_gemma":[0.001041519,0.000004987817,0.0899136,0.003781877,0.000008901572,0.00002023308,0.0001095908,0.1246361,0.7781655,0.0005889305,0.001450186,0.000278603],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6596047,0.0001285888,0.3377615,0.00110631,0.0001796295,0.0002329356,0.000003082492,0.0001302587,0.0008530166],"genre_scores_gemma":[0.9522682,0.0000683173,0.0467756,0.000008133056,0.00001143605,0.00007450113,0.00000258393,0.000008583616,0.0007825767],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.732877,"threshold_uncertainty_score":0.3535452,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.008998518496103433,"score_gpt":0.2160289706201412,"score_spread":0.2070304521240378,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}