{"id":"W4413024604","doi":"10.1002/asi.70013","title":"Understanding discrepancies in the coverage of <scp>OpenAlex</scp> : The case of China","year":2025,"lang":"en","type":"article","venue":"Journal of the Association for Information Science and Technology","topic":"scientometrics and bibliometrics research","field":"Decision Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; Bureau de Coopération Interuniversitaire; Université du Québec à Montréal","funders":"Fonds de recherche du Québec; China Scholarship Council; Social Sciences and Humanities Research Council of Canada; National Natural Science Foundation of China","keywords":"China; Computer science; Data science; Information retrieval; Geography","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":["bibliometrics","metaresearch"],"domain":"evaluation","study_design":"observational","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"gpt","categories":["bibliometrics"],"domain":null,"study_design":"design_other","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"}],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.008182893,0.000186118,0.0005720647,0.01274645,0.0009378354,0.003342154,0.001006544,0.0003733046,0.003403882],"category_scores_gemma":[0.04045862,0.0001499074,0.0003930679,0.02555132,0.0009247736,0.002631708,0.002430371,0.0004351456,0.0003733684],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002764216,"about_ca_system_score_gemma":0.002977037,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.05479484,"about_ca_topic_score_gemma":0.04107728,"domain_scores_codex":[0.994972,0.0009244051,0.0006697636,0.0006484484,0.001937059,0.0008483003],"domain_scores_gemma":[0.9500362,0.01985763,0.01234198,0.005900084,0.01054388,0.001320251],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001167553,0.00003494397,0.9026832,0.0005019851,0.0002803716,0.000949252,0.004567245,0.002634925,0.0007969938,0.01995211,0.01013258,0.05734959],"study_design_scores_gemma":[0.00001120225,0.00002629542,0.9670167,0.0002036402,0.0001137116,0.0002518838,0.003001421,0.002464432,0.001229778,0.002438724,0.02321679,0.00002529141],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9628134,0.001977556,0.0009976813,0.002579037,0.00002764039,0.00002859756,0.01068497,0.00008045925,0.02081065],"genre_scores_gemma":[0.9937598,0.0004479843,0.0002997442,0.0001156694,0.00002909639,0.00001560888,0.004495695,0.00001569447,0.0008206238],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9918171,"threshold_uncertainty_score":0.1089518,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2534551526371789,"score_gpt":0.4762531204382082,"score_spread":0.2227979678010293,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}