{"id":"W2913028816","doi":"","title":"Proceedings of the third international workshop on Large scale testing","year":2014,"lang":"en","type":"article","venue":"","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"","keywords":"Computer science; Acceptance testing; Unit testing; Schedule; Software performance testing; Scale (ratio); Integration testing; Software engineering; Software testing; System testing; Software; Software system; Operating system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004975976,0.00005299106,0.0000678824,0.00002581329,0.00007250763,0.00004415429,0.0007671287,0.00003311719,0.000006366774],"category_scores_gemma":[0.000272105,0.00002839381,0.00003829861,0.0002445679,0.00002321321,0.0001936961,0.0002286849,0.00008005809,0.00002351006],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000156486,"about_ca_system_score_gemma":0.00001328517,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000005233696,"about_ca_topic_score_gemma":0.000001978918,"domain_scores_codex":[0.9993336,0.000006699929,0.0001432074,0.0001595726,0.000246416,0.0001104717],"domain_scores_gemma":[0.9994406,0.0001164954,0.0000745324,0.0001864739,0.0001613735,0.00002050686],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000007119807,0.0001929676,0.8704218,0.00006946844,0.00001372528,8.093868e-8,0.001659265,0.0001155464,0.001161621,0.07419622,0.007720356,0.04444182],"study_design_scores_gemma":[0.0007779843,0.0001365611,0.5726298,0.0005577815,0.000005652701,0.00001505418,0.0003070887,0.3786229,0.01964366,0.006760287,0.02022023,0.0003229181],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7917808,0.000003362681,0.04445263,0.001890086,0.001072256,0.00015038,4.882177e-7,0.000184451,0.1604655],"genre_scores_gemma":[0.9898755,3.55469e-7,0.008854593,0.0003394693,0.00008410186,0.000004672596,6.938284e-8,0.000002155198,0.0008391126],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3785074,"threshold_uncertainty_score":0.1425529,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01366349227969791,"score_gpt":0.2378753392185736,"score_spread":0.2242118469388757,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}