{"id":"W4297779989","doi":"10.5281/zenodo.7069915","title":"D3: A Massive Dataset of Scholarly Metadata for Analyzing the State of Computer Science Research","year":2022,"lang":"en","type":"paratext","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Metadata; Computer science; Data science; State (computer science); Information retrieval; World Wide Web; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.00154219,0.0008656745,0.0007093556,0.01267521,0.0009997498,0.003088213,0.001343328,0.00171379,0.01933313],"category_scores_gemma":[0.009859269,0.0004051964,0.0006675196,0.02103077,0.000521968,0.002877508,0.003322014,0.001161652,0.0282882],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001360051,"about_ca_system_score_gemma":0.002404129,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01088685,"about_ca_topic_score_gemma":0.01896802,"domain_scores_codex":[0.9978176,0.0002968817,0.0003029725,0.0003182916,0.001084532,0.0001795895],"domain_scores_gemma":[0.9927914,0.001867016,0.0005770901,0.0019767,0.001851027,0.0009368187],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002547124,0.00008366825,0.004993074,0.001623958,0.0000665506,0.0002698131,0.0003721931,0.001452924,0.003480955,0.005683709,0.9386179,0.04310066],"study_design_scores_gemma":[0.00007636985,0.00003346753,0.0182854,0.0002130193,0.00003625294,0.0002015542,0.0005746818,0.001356018,0.0031245,0.005157181,0.9708844,0.00005723985],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.004336244,0.0004694043,0.002061056,0.0005022457,0.0001137984,0.0000825682,0.9806182,0.002440923,0.009375533],"genre_scores_gemma":[0.006551618,0.0003782033,0.004593067,0.00008935473,0.00005255925,0.0001172801,0.9844905,0.0003276934,0.00339981],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.9984578,"threshold_uncertainty_score":0.06467575,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09868910898531755,"score_gpt":0.3285302291203945,"score_spread":0.2298411201350769,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}