{"id":"W4224435188","doi":"10.5281/zenodo.6477785","title":"D3: A Massive Dataset of Scholarly Metadata for Analyzing the State of Computer Science Research","year":2022,"lang":"en","type":"paratext","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Metadata; Computer science; Data science; State (computer science); Information retrieval; World Wide Web; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","sts","scholarly_communication","open_science","insufficient_payload"],"consensus_categories":["metaresearch","open_science","insufficient_payload"],"category_scores_codex":[0.05238984,0.0002181853,0.0004738618,0.002407143,0.006028383,0.009906831,0.01539794,0.00005013109,0.02361023],"category_scores_gemma":[0.01263275,0.0001629783,0.00013557,0.007051974,0.001796274,0.001848018,0.02333826,0.0008607432,0.005739217],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002204924,"about_ca_system_score_gemma":0.00008941213,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00008449607,"about_ca_topic_score_gemma":0.000001120896,"domain_scores_codex":[0.9892299,0.002161272,0.00119899,0.001676387,0.005023019,0.0007104576],"domain_scores_gemma":[0.9878635,0.001190016,0.0009630344,0.004020554,0.005764057,0.0001988273],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00006010445,0.00008433944,3.851794e-7,0.0000588034,0.0000579133,0.000002240304,0.000495547,0.001474747,0.0004660656,0.0008157933,0.8407195,0.1557646],"study_design_scores_gemma":[0.0002788275,0.0002976124,0.00003763908,0.00004845425,0.0000248252,0.000007010508,0.0005945889,0.007073749,0.0002800424,0.0006631574,0.9905343,0.0001597751],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"dataset","genre_scores_codex":[0.007077907,0.001757039,0.5862312,0.004642948,0.005645855,0.007024304,0.3112153,0.0002939896,0.07611144],"genre_scores_gemma":[0.2556342,0.001738904,0.03513777,0.001070969,0.002523551,0.000004802825,0.3803163,0.007064985,0.3165085],"genre_candidate":"dataset","genre_consensus":null,"teacher_disagreement_score":0.5510934,"threshold_uncertainty_score":0.9956843,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.242958847849084,"score_gpt":0.4077420139333628,"score_spread":0.1647831660842788,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}