{"id":"W4393528085","doi":"10.5281/zenodo.7708984","title":"Defectors: A Large Scale Python Dataset for Defect Prediction","year":2023,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Industrial Vision Systems and Defect Detection","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Python (programming language); Computer science; Scale (ratio); Programming language; Cartography; Geography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001166732,0.002086789,0.0007510447,0.003393648,0.0008317517,0.0008402118,0.002702154,0.001463201,0.005086905],"category_scores_gemma":[0.006023011,0.0005261063,0.001260741,0.003907102,0.0007264349,0.001718935,0.002266576,0.002144946,0.008354676],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001120033,"about_ca_system_score_gemma":0.00225103,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01874437,"about_ca_topic_score_gemma":0.03246696,"domain_scores_codex":[0.9981402,0.0001953861,0.000193116,0.0004211663,0.0008121406,0.0002380556],"domain_scores_gemma":[0.997053,0.0004666339,0.0003735517,0.0008274751,0.0008809657,0.0003983291],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0005995805,0.0004956182,0.02430437,0.0008462863,0.0001570678,0.0005327364,0.0001594637,0.008145408,0.003062917,0.001380404,0.9167987,0.04351743],"study_design_scores_gemma":[0.0009986882,0.000857103,0.1645549,0.0004712747,0.0002057583,0.001765369,0.0006013081,0.1209303,0.01782878,0.01084524,0.6805252,0.0004160415],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.03464137,0.0004092384,0.004839677,0.0005833697,0.0001798132,0.0002749043,0.9364771,0.01920089,0.003393657],"genre_scores_gemma":[0.02264159,0.0001216239,0.006269427,0.0001417037,0.0000252239,0.0002797499,0.9689313,0.000462323,0.001127111],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.01874437,"threshold_uncertainty_score":0.03727055,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03739181245637486,"score_gpt":0.2548971986705139,"score_spread":0.2175053862141391,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}