{"id":"W4400266869","doi":"10.1145/3643991.3644907","title":"PeaTMOSS: A Dataset and Initial Analysis of Pre-Trained Models in Open-Source Software","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"National Science Foundation","keywords":"Computer science; Open source software; Open source; Software; Artificial intelligence; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00209554,0.001791034,0.0006373668,0.002937668,0.0008781599,0.001219792,0.002426606,0.002206628,0.003942556],"category_scores_gemma":[0.01236333,0.0005849344,0.001620438,0.002285341,0.001009554,0.001964272,0.002141502,0.00254864,0.005146643],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001561277,"about_ca_system_score_gemma":0.002032148,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02539775,"about_ca_topic_score_gemma":0.0638492,"domain_scores_codex":[0.9981964,0.000354686,0.0001790015,0.0004721762,0.0006006497,0.0001971088],"domain_scores_gemma":[0.9939619,0.002488657,0.0003762971,0.001604211,0.001221123,0.0003477762],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00160612,0.001379587,0.06892976,0.002770936,0.0007159352,0.001975193,0.0009971659,0.1019496,0.0092711,0.006822267,0.635078,0.1685045],"study_design_scores_gemma":[0.0008084566,0.00118994,0.1168041,0.001030511,0.0003287176,0.001541742,0.001426495,0.3955546,0.02519174,0.02413718,0.4315754,0.0004111267],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.3472562,0.003724356,0.0379922,0.002482716,0.001071678,0.0004735089,0.5408654,0.05053747,0.01559667],"genre_scores_gemma":[0.1881986,0.000957053,0.03229921,0.0004695131,0.0001122294,0.0006092572,0.7695743,0.002476199,0.005303614],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.02539775,"threshold_uncertainty_score":0.05049986,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03556240376042519,"score_gpt":0.3328252172419072,"score_spread":0.297262813481482,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}