{"id":"W4391506236","doi":"10.48550/arxiv.2402.00699","title":"PeaTMOSS: A Dataset and Initial Analysis of Pre-Trained Models in Open-Source Software","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Argonne National Laboratory; Purdue University; Cisco Systems; U.S. Department of Energy; Office of Science; National Science Foundation","keywords":"Open source; Open source software; Computer science; Software; Artificial intelligence; Statistics; Data mining; Machine learning; Mathematics; Operating system","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002510803,0.00166966,0.0006425682,0.004621147,0.0007920137,0.001389727,0.002482416,0.001689027,0.002759803],"category_scores_gemma":[0.01567611,0.0006418096,0.001632617,0.004904707,0.0008473538,0.002825839,0.002433936,0.002046495,0.003947075],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001670735,"about_ca_system_score_gemma":0.001860037,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0205734,"about_ca_topic_score_gemma":0.04027099,"domain_scores_codex":[0.9975272,0.0004533418,0.0003369951,0.0006866371,0.0007872552,0.0002085966],"domain_scores_gemma":[0.9922484,0.003234381,0.0006408749,0.002182971,0.001392394,0.0003009868],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"observational","study_design_scores_codex":[0.0009029285,0.0007681435,0.09037274,0.003374652,0.000472603,0.001710567,0.001286161,0.05155038,0.006972691,0.006223205,0.6379388,0.1984272],"study_design_scores_gemma":[0.0004706941,0.0006262758,0.1570941,0.001038414,0.0002442784,0.001590343,0.001421459,0.2451422,0.01766892,0.01595688,0.5584152,0.0003312122],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.2455838,0.002903965,0.02508544,0.001731178,0.0003958844,0.0003507211,0.6665494,0.04934246,0.008057064],"genre_scores_gemma":[0.08903615,0.0006461155,0.02693816,0.0003078566,0.00005133402,0.0005130011,0.8782372,0.001999375,0.002270791],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.0205734,"threshold_uncertainty_score":0.04090732,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2667119319386875,"score_gpt":0.3215716429832131,"score_spread":0.05485971104452553,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}