{"id":"W4406367101","doi":"10.1002/asi.24979","title":"Evaluating the linguistic coverage of <scp>OpenAlex</scp>: An assessment of metadata accuracy and completeness","year":2025,"lang":"en","type":"article","venue":"Journal of the Association for Information Science and Technology","topic":"scientometrics and bibliometrics research","field":"Decision Sciences","cited_by":44,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université du Québec à Montréal; Lakehead University; Université de Montréal","funders":"Social Sciences and Humanities Research Council of Canada; Fonds de Recherche du Québec-Société et Culture; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung","keywords":"Metadata; Computer science; Scopus; Publishing; English language; Completeness (order theory); World Wide Web; Quality (philosophy); Information retrieval; Library science; Linguistics; MEDLINE; Mathematics; Political science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.083191,0.0004747914,0.001037993,0.03383928,0.001977878,0.007000898,0.001425529,0.0008868893,0.003851212],"category_scores_gemma":[0.31075,0.0002936801,0.001020995,0.02765731,0.003766775,0.004989012,0.009023609,0.000632724,0.001393914],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002147583,"about_ca_system_score_gemma":0.003569952,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004994245,"about_ca_topic_score_gemma":0.005348494,"domain_scores_codex":[0.9110028,0.02716865,0.01685227,0.005180648,0.03845717,0.001338461],"domain_scores_gemma":[0.5310655,0.2681438,0.05682779,0.03855127,0.1031621,0.002249619],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0006559034,0.0002082017,0.6625879,0.005697183,0.00102181,0.0006655037,0.03773091,0.002526215,0.005299568,0.01795251,0.01607959,0.2495747],"study_design_scores_gemma":[0.00006675912,0.00033551,0.8026175,0.005614898,0.0009324427,0.001113613,0.04381974,0.01125672,0.01644806,0.01229037,0.1052637,0.0002406788],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9220532,0.003599728,0.02021872,0.002884824,0.0003205979,0.0005375385,0.01117259,0.00040097,0.03881192],"genre_scores_gemma":[0.97535,0.001074444,0.01056787,0.0002822911,0.0001521166,0.0004487241,0.009997116,0.0002152333,0.001912324],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9661607,"threshold_uncertainty_score":0.4399613,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3736602032162077,"score_gpt":0.602059430984761,"score_spread":0.2283992277685533,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}