{"id":"W4393488795","doi":"10.5281/zenodo.2558377","title":"SeSaMe: A Data Set of Semantically Similar Java Methods","year":2019,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Java; Computer science; Set (abstract data type); Programming language; Information retrieval","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001775909,0.00213233,0.001257758,0.007898723,0.001483547,0.002494839,0.003038242,0.002562156,0.008828196],"category_scores_gemma":[0.01007993,0.0007292201,0.002379297,0.008286491,0.0008004016,0.002673938,0.003575228,0.002971007,0.01236087],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001585168,"about_ca_system_score_gemma":0.003288478,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01712937,"about_ca_topic_score_gemma":0.02809485,"domain_scores_codex":[0.9959223,0.000497674,0.0008583556,0.001003966,0.001328261,0.0003893604],"domain_scores_gemma":[0.9943528,0.001507867,0.0006806374,0.001488166,0.001455719,0.0005147828],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001000801,0.0004219316,0.0106258,0.004510287,0.000228947,0.0004603624,0.0004002048,0.001806749,0.004599612,0.006587388,0.9405999,0.02875811],"study_design_scores_gemma":[0.0004805164,0.00009965931,0.0159255,0.0005464326,0.00009202311,0.0005786447,0.0004135278,0.002011966,0.004126919,0.003591451,0.9720433,0.0000900601],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.003566567,0.0003432013,0.001260587,0.0002080034,0.00009843193,0.0001131697,0.9912528,0.001532612,0.001624658],"genre_scores_gemma":[0.002883763,0.000122877,0.002608755,0.00009612089,0.000011091,0.000257912,0.9934525,0.0001424404,0.0004244417],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.01712937,"threshold_uncertainty_score":0.03405929,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.410363912626614,"score_gpt":0.4676542377617528,"score_spread":0.05729032513513882,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}