{"id":"W6912674579","doi":"10.5281/zenodo.6477784","title":"D3: A Massive Dataset of Scholarly Metadata for Analyzing the State of Computer Science Research","year":2022,"lang":"en","type":"article","venue":"NPARC","topic":"Research Data Management Practices","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Metadata; State (computer science); Meta Data Services; Data element; Key (lock)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.003291584,0.0007578968,0.0008454907,0.01902466,0.001295901,0.003732905,0.001594791,0.001552749,0.02088499],"category_scores_gemma":[0.01958863,0.0005415724,0.0006904178,0.02865116,0.0005832529,0.003122571,0.004523681,0.001528481,0.02904398],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001673229,"about_ca_system_score_gemma":0.004089283,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01345557,"about_ca_topic_score_gemma":0.02468483,"domain_scores_codex":[0.9958373,0.0005172525,0.0006899078,0.0005371921,0.002139377,0.0002789285],"domain_scores_gemma":[0.9789106,0.005317906,0.002162176,0.005406268,0.005563835,0.002639312],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002352478,0.00008361942,0.008294132,0.001596788,0.00007098908,0.0001947149,0.0006019315,0.0007163192,0.003172801,0.004534757,0.9323424,0.04815628],"study_design_scores_gemma":[0.00008508642,0.00004043018,0.02896446,0.0003379284,0.00004100374,0.0001769097,0.000880155,0.0008560077,0.002705726,0.004373553,0.9614699,0.00006881249],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.00429153,0.0005616717,0.002261155,0.0009294419,0.0001638714,0.0001407979,0.9787853,0.002745993,0.01012024],"genre_scores_gemma":[0.007414175,0.0005528514,0.008359124,0.0001509421,0.00009665684,0.0001939399,0.9793865,0.00044398,0.003401826],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.9967084,"threshold_uncertainty_score":0.06986725,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2186397450307563,"score_gpt":0.4384354399892839,"score_spread":0.2197956949585277,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}