{"id":"W4393508681","doi":"10.5281/zenodo.7570822","title":"Defectors: A Large Scale Python Dataset for Defect Prediction","year":2023,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Industrial Vision Systems and Defect Detection","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Python (programming language); Computer science; Scale (ratio); Programming language; Cartography; Geography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.001409846,0.000342685,0.0003695448,0.0005980895,0.001701219,0.0008183089,0.00100816,0.0004537326,0.001560857],"category_scores_gemma":[0.0009417612,0.0003713684,0.000198277,0.000774207,0.00004611923,0.0002351515,0.0007779473,0.000625322,0.02350618],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003192578,"about_ca_system_score_gemma":0.000005113873,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005982405,"about_ca_topic_score_gemma":0.000006588478,"domain_scores_codex":[0.997525,0.0002789358,0.0005061812,0.0006084114,0.0004755428,0.0006058743],"domain_scores_gemma":[0.9981493,0.00007579029,0.0001388911,0.001129344,0.0003021636,0.0002045272],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000052946,0.00004366294,1.0267e-7,0.0004309254,0.0001069128,0.000006343989,0.00006816685,0.0003271323,0.0002013473,0.000005079142,0.9957622,0.002995138],"study_design_scores_gemma":[0.0008060776,0.000319678,0.00001246582,0.0001412785,0.00009128181,0.00006094572,0.00008464844,0.001212374,0.0000859865,0.00001750631,0.9968403,0.0003274684],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.0001745398,0.0000750765,0.003041662,0.00001922983,0.001510057,0.001272353,0.9912428,0.00213698,0.0005272985],"genre_scores_gemma":[0.0004280792,0.0001435538,0.00003198106,0.00002262513,0.00119022,0.000001016924,0.9958959,0.002147272,0.0001393894],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.02194532,"threshold_uncertainty_score":0.9998738,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03739181245637486,"score_gpt":0.2548971986705139,"score_spread":0.2175053862141391,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}