{"id":"W4413910953","doi":"10.1038/s41597-025-05445-3","title":"The Indo-European Cognate Relationships dataset","year":2025,"lang":"en","type":"article","venue":"Scientific Data","topic":"Language and cultural evolution","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Lexeme; Cognate; Metadata; Computer science; Benchmark (surveying); Lexicon; Language family; Natural language processing; Interoperability; Linguistics; Artificial intelligence; Information retrieval; World Wide Web; Geography; Cartography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001070113,0.0009896707,0.0007086881,0.00573887,0.001196882,0.001882344,0.00204323,0.001470185,0.01665124],"category_scores_gemma":[0.005460748,0.0002575071,0.0008415338,0.0085527,0.000555508,0.001386466,0.002899858,0.001205401,0.015343],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001063549,"about_ca_system_score_gemma":0.001699564,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01140051,"about_ca_topic_score_gemma":0.01991587,"domain_scores_codex":[0.9986864,0.0002503345,0.0001878342,0.0004053141,0.0003021828,0.0001679934],"domain_scores_gemma":[0.9984542,0.0004697503,0.0001798741,0.0003535613,0.0003748892,0.0001677312],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0003390065,0.0001675667,0.02046454,0.002301765,0.0001448579,0.0007129987,0.0009633527,0.001364052,0.002433957,0.01153462,0.9277706,0.03180275],"study_design_scores_gemma":[0.00006431383,0.00002184442,0.02382772,0.0003369918,0.00004510648,0.0003621772,0.0006859989,0.0007840388,0.0009064771,0.002233851,0.9706898,0.00004174812],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.008432833,0.0004179194,0.0007482895,0.0002096081,0.00004276811,0.00005470651,0.9835434,0.0004268399,0.00612365],"genre_scores_gemma":[0.004935592,0.0001014178,0.001490068,0.0000728086,0.00001208206,0.0001358347,0.9922181,0.00006485329,0.0009691172],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.01665124,"threshold_uncertainty_score":0.05570394,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1068008655528147,"score_gpt":0.3751298188663035,"score_spread":0.2683289533134888,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}