{"id":"W4319811928","doi":"10.22541/au.167528154.45763422/v1","title":"D3: A Massive Dataset of Scholarly Metadata for Analyzing the State of Computer Science Research","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Metadata; Computer science; Citation; Data science; Field (mathematics); Information retrieval; World Wide Web; Bibliometrics; Library science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.003379939,0.00113572,0.00098087,0.03042044,0.001554177,0.003751878,0.002021673,0.001951898,0.01337927],"category_scores_gemma":[0.02238179,0.0005486197,0.0008529433,0.0432531,0.0006906043,0.002841214,0.003871483,0.00148083,0.01611078],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002361916,"about_ca_system_score_gemma":0.005135017,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01270245,"about_ca_topic_score_gemma":0.02594349,"domain_scores_codex":[0.9949473,0.0007757891,0.001035835,0.0007752762,0.001992194,0.0004735346],"domain_scores_gemma":[0.9772538,0.008312413,0.003564071,0.003543818,0.005364994,0.001961034],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0003302441,0.0002103189,0.02777577,0.004727847,0.000182819,0.0005115428,0.001119871,0.002000328,0.005779696,0.009866544,0.8905881,0.05690685],"study_design_scores_gemma":[0.000129247,0.00005827973,0.03401868,0.0004836505,0.00007518718,0.0003235894,0.0009306603,0.001713295,0.003520847,0.005236965,0.9534222,0.00008731588],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.006509664,0.0009506373,0.001866887,0.0006665524,0.0001195983,0.0001027904,0.983481,0.001094897,0.005207873],"genre_scores_gemma":[0.007792508,0.0007077756,0.006124736,0.0001460975,0.00009714274,0.0002754784,0.9830548,0.0001531131,0.001648235],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.9966201,"threshold_uncertainty_score":0.04475814,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6048854509862936,"score_gpt":0.5447832631191812,"score_spread":0.06010218786711241,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}