{"meta":{"query_hash":"8953a914060a","filters":{"venue":"Data Intelligence"},"cohort_total":12,"direct_labels_cover":0,"predictions_cover":12,"exported":12,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/8953a914060a","api":"https://metacan.xera.ac/api/v1/cohort?venue=Data+Intelligence"},"results":[{"id":"W2985716559","doi":"10.1162/dint_a_00039","title":"The FAIR Funding Model: Providing a Framework for Research Funders to Drive the Transition toward FAIR Data Management and Stewardship Practices","year":2019,"lang":"en","type":"article","venue":"Data Intelligence","topic":"Research Data Management Practices","field":"Computer Science","cited_by":42,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Canadian Institutes of Health Research; Stroke Association","keywords":"RDM; Stewardship (theology); Metadata; Business; Work (physics); Process management; Knowledge management; Data curation; Public relations; Computer science; Political science; Data science; Engineering; World Wide Web","score_opus":0.6366391164030468,"score_gpt":0.5222957442909713,"score_spread":0.11434337211207557,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2985716559","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0034357759,0.000800101,0.7928189,0.14142162,0.0010671712,0.0021160038,0.00041785729,0.001582696,0.056339867],"genre_scores_gemma":[0.086119026,0.0008548859,0.88476825,0.01056387,0.00077282457,0.004725689,0.00052182033,0.00060129666,0.011072313],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.7172445,0.20833485,0.016591864,0.015834784,0.033508725,0.008485205],"domain_scores_gemma":[0.6896175,0.16142176,0.019624025,0.061492294,0.04596932,0.021875056],"candidate_categories":["metaresearch","open_science"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.33797568,0.0021970596,0.0024528904,0.009735446,0.011612217,0.047587734,0.012795637,0.01699477,0.011138635],"category_scores_gemma":[0.30561528,0.0025040538,0.0037447996,0.009477737,0.034740835,0.053251255,0.03115489,0.016850166,0.004909509],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000030834522,0.00005760285,0.00079810666,0.0001552374,0.00003042099,0.000080401725,0.0025549706,0.0025562586,0.00011571888,0.9649222,0.010567903,0.018130383],"study_design_scores_gemma":[0.000077127224,0.000049693186,0.00022078711,0.0005067424,0.000023260693,0.00006508218,0.0011288199,0.0059666894,0.00022838927,0.8932176,0.0984384,0.00007736185],"about_ca_topic_score_codex":0.01830476,"about_ca_topic_score_gemma":0.016585741,"teacher_disagreement_score":0.9872044,"about_ca_system_score_codex":0.029464956,"about_ca_system_score_gemma":0.11377686,"threshold_uncertainty_score":0.81639385},"labels":[],"label_agreement":null},{"id":"W2987816848","doi":"10.1162/dint_a_00051","title":"Considerations for the Conduction and Interpretation of FAIRness Evaluations","year":2019,"lang":"en","type":"article","venue":"Data Intelligence","topic":"Research Data Management Practices","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"National Institutes of Health; Horizon 2020 Framework Programme; Nederlandse Organisatie voor Wetenschappelijk Onderzoek","keywords":"Metadata; Interpretation (philosophy); Computer science; Maturity (psychological); Software; Psychology; World Wide Web","score_opus":0.33447316204815425,"score_gpt":0.4639573840777451,"score_spread":0.12948422202959087,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2987816848","genre_codex":"methods","genre_gemma":"methods","domain_codex":"methods","domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019554542,0.007919106,0.7202725,0.1351521,0.009583512,0.012830004,0.0005767668,0.00097509514,0.09313647],"genre_scores_gemma":[0.31105316,0.0015940875,0.631912,0.019136248,0.0035637028,0.025642643,0.00028419375,0.000930617,0.0058833747],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.18368275,0.69470435,0.04567996,0.017581102,0.052064683,0.0062871706],"domain_scores_gemma":[0.12603807,0.6633445,0.023668922,0.062100913,0.119683795,0.0051638014],"candidate_categories":["metaresearch","open_science"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.7534164,0.0025106466,0.004506251,0.009302016,0.011389375,0.027089197,0.009554322,0.011340588,0.0068340194],"category_scores_gemma":[0.8436308,0.0018727384,0.0047359644,0.00714169,0.0324056,0.02042385,0.012248508,0.020984078,0.0020977347],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00086093525,0.0002578194,0.0047626803,0.0019744367,0.00045409094,0.00027977533,0.03992791,0.0022579944,0.000990544,0.778668,0.029503351,0.14006251],"study_design_scores_gemma":[0.00052967615,0.00041099137,0.0049095843,0.00941769,0.00028635003,0.00034104532,0.010629246,0.009206367,0.0029451032,0.83966714,0.12125255,0.0004043466],"about_ca_topic_score_codex":0.0070327497,"about_ca_topic_score_gemma":0.0069941953,"teacher_disagreement_score":0.9904457,"about_ca_system_score_codex":0.01658952,"about_ca_system_score_gemma":0.041869365,"threshold_uncertainty_score":0.3040815},"labels":[],"label_agreement":null},{"id":"W2988604061","doi":"10.1162/dint_a_00029","title":"Ontology-based Access Control for FAIR Data","year":2019,"lang":"en","type":"article","venue":"Data Intelligence","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Metadata; Computer science; Ontology; Access control; Data access; Metadata repository; Database; World Wide Web; Information retrieval; Computer security","score_opus":0.5944593014190421,"score_gpt":0.5293744523263445,"score_spread":0.06508484909269763,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2988604061","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0048001404,0.00022468108,0.98554,0.0021588625,0.0001267092,0.0002669625,0.00011906304,0.00033323167,0.006430389],"genre_scores_gemma":[0.26827726,0.0007533256,0.72144604,0.001144111,0.000373756,0.001144952,0.0005953919,0.00022896868,0.0060361726],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.95629716,0.01394186,0.0053463215,0.005115385,0.01683077,0.0024685413],"domain_scores_gemma":[0.95485973,0.016286768,0.002992721,0.018161628,0.005875212,0.0018238323],"candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.038779747,0.0008555293,0.0016593254,0.003665622,0.005131209,0.014298766,0.0043120976,0.0031346597,0.0023349405],"category_scores_gemma":[0.044618532,0.0009875661,0.0036961667,0.003602441,0.013833791,0.020987868,0.011299847,0.008257696,0.000634227],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000022013393,0.000051883628,0.00023041619,0.000059001643,0.00003705351,0.000071051836,0.0006530367,0.004774726,0.0008681332,0.98057634,0.000875943,0.011780371],"study_design_scores_gemma":[0.00003909111,0.000031313626,0.00018226699,0.000096660035,0.0000542569,0.00013218842,0.00023848849,0.043748625,0.002565355,0.9205528,0.032291926,0.00006707159],"about_ca_topic_score_codex":0.011365454,"about_ca_topic_score_gemma":0.0055232793,"teacher_disagreement_score":0.9956879,"about_ca_system_score_codex":0.0063845837,"about_ca_system_score_gemma":0.010427432,"threshold_uncertainty_score":0.20508939},"labels":[],"label_agreement":null},{"id":"W2988998130","doi":"10.1162/dint_r_00024","title":"FAIR Principles: Interpretations and Implementation Considerations","year":2019,"lang":"en","type":"article","venue":"Data Intelligence","topic":"Research Data Management Practices","field":"Computer Science","cited_by":459,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Biotechnology and Biological Sciences Research Council; European Regional Development Fund; Horizon 2020 Framework Programme; Innovative Medicines Initiative; National Institutes of Health; National Science Foundation; Ministerio de Economía y Competitividad; Common Fund; Nederlandse Organisatie voor Wetenschappelijk Onderzoek; European Commission; Universidad Politécnica de Madrid","keywords":"Implementation; Interoperability; Computer science; Reusability; USable; Fair use; Reuse; Stakeholder; Convergence (economics); Risk analysis (engineering); Data science; Management science; Software engineering; World Wide Web; Business; Engineering; Political science; Public relations; Software; Law; Economics","score_opus":0.21350192087230005,"score_gpt":0.4376810240360997,"score_spread":0.22417910316379966,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2988998130","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008199487,0.004203968,0.51284593,0.27687195,0.003124818,0.0014681682,0.00024069918,0.0004327564,0.19261225],"genre_scores_gemma":[0.40430656,0.0031331163,0.5151475,0.045444496,0.0019639467,0.004661476,0.00028508948,0.0005973769,0.024460483],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.76982343,0.13769525,0.016470933,0.013843876,0.053240947,0.008925581],"domain_scores_gemma":[0.81024706,0.12227935,0.0068575013,0.025198814,0.031102737,0.004314461],"candidate_categories":["metaresearch","open_science"],"consensus_categories":[],"category_scores_codex":[0.20466621,0.0020990246,0.002674701,0.006076955,0.014409608,0.030912865,0.0097557,0.021796219,0.008510953],"category_scores_gemma":[0.23214829,0.0019808137,0.0029800565,0.004311164,0.09125117,0.034942966,0.020422181,0.026193425,0.0019733375],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000007138813,0.000015369935,0.00008901838,0.0000657216,0.0000092614055,0.000042152333,0.00200769,0.0002873132,0.000045902514,0.9910625,0.0022876905,0.004080202],"study_design_scores_gemma":[0.000015770218,0.000015173389,0.00008358962,0.00035723113,0.000013259405,0.000049337912,0.0014441428,0.00062318554,0.00017545206,0.95857286,0.03861935,0.00003065089],"about_ca_topic_score_codex":0.0099395765,"about_ca_topic_score_gemma":0.0075395405,"teacher_disagreement_score":0.9902443,"about_ca_system_score_codex":0.016725613,"about_ca_system_score_gemma":0.032715417,"threshold_uncertainty_score":0.9807882},"labels":[],"label_agreement":null},{"id":"W3017715546","doi":"10.1162/dint_a_00058","title":"The Semantic Data Dictionary – An Approach for Describing and Annotating Data","year":2020,"lang":"en","type":"article","venue":"Data Intelligence","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":49,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"CARE Canada; Privacy Analytics (Canada)","funders":"National Institute of Environmental Health Sciences","keywords":"Computer science; Natural language processing; Data dictionary; Information retrieval; Artificial intelligence; World Wide Web; Metadata","score_opus":0.3725294609797752,"score_gpt":0.37406425153345557,"score_spread":0.0015347905536803874,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3017715546","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001125762,0.00050260714,0.98273164,0.002434977,0.0003308161,0.0007745255,0.0046403883,0.001840637,0.0056187618],"genre_scores_gemma":[0.013279613,0.0009341579,0.96846306,0.001407623,0.00014898027,0.0013282051,0.011383435,0.0009005755,0.002154369],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9608474,0.01607397,0.010315718,0.0047968104,0.0071904142,0.0007756839],"domain_scores_gemma":[0.92899406,0.02610219,0.0049694953,0.027786447,0.010478642,0.0016692375],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.034431316,0.0018237968,0.0024186429,0.020726489,0.005183016,0.013950745,0.006920692,0.0041535487,0.005078309],"category_scores_gemma":[0.06554477,0.0022999144,0.0043082377,0.024620656,0.008127064,0.028139997,0.0157383,0.009992349,0.0041063046],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016435627,0.00013103274,0.0023987575,0.0018497347,0.00023705387,0.00059153925,0.008423929,0.0047395816,0.004845076,0.78626823,0.040177904,0.15017271],"study_design_scores_gemma":[0.000041396765,0.000062125306,0.00070492283,0.0016058718,0.00011163087,0.00076965534,0.0033589923,0.013745306,0.004673659,0.32289666,0.6518505,0.00017921776],"about_ca_topic_score_codex":0.013728149,"about_ca_topic_score_gemma":0.013377624,"teacher_disagreement_score":0.034431316,"about_ca_system_score_codex":0.004709535,"about_ca_system_score_gemma":0.016220227,"threshold_uncertainty_score":0.18209243},"labels":[],"label_agreement":null},{"id":"W3128327914","doi":"10.1162/dint_a_00086","title":"Implementation of an Open Science Instruction Program for Undergraduates","year":2021,"lang":"en","type":"article","venue":"Data Intelligence","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Okanagan University College; University of British Columbia, Okanagan Campus; University of British Columbia","funders":"","keywords":"Transparency (behavior); Curriculum; Context (archaeology); Best practice; Political science; Undergraduate research; Subject (documents); Medical education; Public relations; Engineering ethics; Pedagogy; Engineering; Library science; Sociology; Computer science; Medicine; Geography","score_opus":0.4297001939532424,"score_gpt":0.566531096562857,"score_spread":0.13683090260961467,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3128327914","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.89471394,0.00019984612,0.016531423,0.026268896,0.0012526158,0.010720696,0.0007392336,0.0016692044,0.047904145],"genre_scores_gemma":[0.7766563,0.00035096172,0.114808254,0.010144056,0.0004706082,0.017088026,0.0021432736,0.00019975999,0.078138754],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9953277,0.0011713847,0.00023384945,0.0005080645,0.0013695109,0.0013895117],"domain_scores_gemma":[0.94277424,0.003482549,0.0017199018,0.0027164195,0.00787908,0.04142796],"candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.0109985145,0.00050215505,0.0005311668,0.0015319815,0.005169207,0.003784028,0.003605088,0.0018534121,0.014276377],"category_scores_gemma":[0.021501444,0.00064050127,0.00054214336,0.00087484316,0.0016580651,0.00226237,0.01059174,0.0046066917,0.0035224643],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00065560243,0.040732197,0.053836867,0.0006968809,0.000026159378,0.0009180214,0.037817743,0.0012122252,0.013589303,0.008004399,0.08582068,0.7566899],"study_design_scores_gemma":[0.0010148193,0.018034974,0.31429812,0.0011594143,0.00006400687,0.0008185733,0.05487792,0.0054849703,0.021344086,0.016329622,0.5662195,0.00035392123],"about_ca_topic_score_codex":0.0039043282,"about_ca_topic_score_gemma":0.014672125,"teacher_disagreement_score":0.99639493,"about_ca_system_score_codex":0.0057889502,"about_ca_system_score_gemma":0.040724367,"threshold_uncertainty_score":0.058166385},"labels":[],"label_agreement":null},{"id":"W4312713146","doi":"10.1162/dint_x_00187","title":"About The Author","year":2022,"lang":"en","type":"article","venue":"Data Intelligence","topic":"Online and Blended Learning","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Political science","score_opus":0.15165752942592567,"score_gpt":0.42420798732970744,"score_spread":0.27255045790378174,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312713146","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0020980586,0.01810216,0.002478784,0.1493497,0.14814317,0.0002101083,0.006232979,0.002108135,0.67127705],"genre_scores_gemma":[0.014727616,0.009135627,0.0011521063,0.051356286,0.010143942,0.00021376372,0.002323696,0.0009934804,0.90995353],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9979679,0.00034152748,0.0001485009,0.0004944712,0.00078242796,0.00026520452],"domain_scores_gemma":[0.9922402,0.0011016239,0.00036528535,0.000662415,0.0039720293,0.001658473],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0013802286,0.0007789258,0.00085962354,0.001471905,0.002790189,0.005939523,0.0016948791,0.0027568475,0.5334479],"category_scores_gemma":[0.018383374,0.0002739154,0.00045347176,0.0014564319,0.0008586895,0.004818907,0.002470889,0.003749686,0.37389687],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000016016907,0.0000102007625,0.00019532716,0.00012238645,0.0000031916597,0.00013316548,0.00028533913,0.000018107887,0.000059119153,0.0029369781,0.9607059,0.035514235],"study_design_scores_gemma":[0.000002014761,0.0000040903187,0.0001080522,0.0001477021,0.0000022145514,0.0002002788,0.0002538003,0.000016367454,0.0000481197,0.0006086405,0.99860424,0.0000044881135],"about_ca_topic_score_codex":0.0025227987,"about_ca_topic_score_gemma":0.0031913833,"teacher_disagreement_score":0.46655208,"about_ca_system_score_codex":0.0021329802,"about_ca_system_score_gemma":0.0044364855,"threshold_uncertainty_score":0.6654799},"labels":[],"label_agreement":null},{"id":"W4380629788","doi":"10.1162/dint_a_00209","title":"MillenniumDB: An Open-Source Graph Database System","year":2023,"lang":"en","type":"article","venue":"Data Intelligence","topic":"Graph Theory and Algorithms","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Agencia Nacional de Investigación y Desarrollo; Pontificia Universidad Católica de Chile; Universidad de Chile; University of Waterloo; University of Toronto; University of Galway; National University of Ireland; Universidad Técnica Federico Santa María","keywords":"Computer science; Graph database; Query language; Query optimization; Graph; Search engine indexing; Information retrieval; Theoretical computer science; Joins; Database; Programming language","score_opus":0.103571274253965,"score_gpt":0.32994977076281207,"score_spread":0.22637849650884706,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4380629788","genre_codex":"software","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":"software","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03052348,0.003176313,0.42391863,0.0020292539,0.0004961025,0.0006481927,0.06270939,0.44815758,0.028340992],"genre_scores_gemma":[0.26994628,0.0027941964,0.48344493,0.0017297684,0.00017963687,0.00070044096,0.20500045,0.02220157,0.014002775],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99841905,0.00022741739,0.0002061473,0.0004075674,0.00063173275,0.00010809791],"domain_scores_gemma":[0.9961966,0.00078245165,0.00022074157,0.0018348064,0.0006414589,0.0003240025],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018188485,0.00087143877,0.0007977335,0.003365005,0.00074562035,0.0039970763,0.005469285,0.0009477075,0.0089140115],"category_scores_gemma":[0.0073566264,0.0006647105,0.00090995245,0.00501042,0.0006926687,0.0065879025,0.0050063203,0.0013219808,0.0042110584],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015897643,0.00042053088,0.0062648403,0.0018692953,0.0005235531,0.0005904809,0.0006131586,0.021646732,0.011098761,0.103154905,0.46437556,0.3878524],"study_design_scores_gemma":[0.00083138334,0.00027474336,0.005197859,0.0003121137,0.00025230122,0.0008957455,0.0008320047,0.2335411,0.030652206,0.14971592,0.5771801,0.0003146057],"about_ca_topic_score_codex":0.01095378,"about_ca_topic_score_gemma":0.011239753,"teacher_disagreement_score":0.01095378,"about_ca_system_score_codex":0.0013605212,"about_ca_system_score_gemma":0.002285185,"threshold_uncertainty_score":0.029820383},"labels":[],"label_agreement":null},{"id":"W4390025222","doi":"10.1162/dint_a_00240","title":"BIKAS: Bio-Inspired Knowledge Acquisition and Simulacrum—A Knowledge Database to Support Multifunctional Design Concept Generation","year":2023,"lang":"en","type":"article","venue":"Data Intelligence","topic":"Design Education and Practice","field":"Engineering","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; McGill University","keywords":"Simulacrum; Computer science; Knowledge acquisition; Knowledge management; Artificial intelligence","score_opus":0.20127910444684868,"score_gpt":0.3756119567879984,"score_spread":0.17433285234114973,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390025222","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01549733,0.00072374224,0.9543192,0.0007099573,0.00008548246,0.0005018736,0.0034933446,0.008995577,0.015673496],"genre_scores_gemma":[0.12092629,0.00087776565,0.8632973,0.00024591497,0.000025221756,0.00054383924,0.007933007,0.00038268723,0.005767941],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988223,0.0002768363,0.00018203203,0.00024777447,0.0004195174,0.00005156964],"domain_scores_gemma":[0.9975889,0.0010130439,0.00019594711,0.00072072103,0.00036760134,0.00011385087],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022210663,0.00070368266,0.0007187353,0.0028972877,0.00059252244,0.0038162468,0.002171931,0.0009714493,0.0069265445],"category_scores_gemma":[0.0070299064,0.0004890101,0.0009723456,0.0016049256,0.00085562957,0.0040688524,0.0029907464,0.0011175835,0.002078464],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005330731,0.00045892288,0.0033816365,0.0014975526,0.00017978689,0.00095449,0.002036012,0.05232885,0.016889669,0.31161988,0.025120644,0.5849995],"study_design_scores_gemma":[0.00018210516,0.00025530983,0.0012947693,0.00078734953,0.0001970032,0.0010864832,0.0009270156,0.4387051,0.03341377,0.15841778,0.36460862,0.00012479963],"about_ca_topic_score_codex":0.0026041362,"about_ca_topic_score_gemma":0.0033789289,"teacher_disagreement_score":0.0069265445,"about_ca_system_score_codex":0.0010358888,"about_ca_system_score_gemma":0.0015996593,"threshold_uncertainty_score":0.023171544},"labels":[],"label_agreement":null},{"id":"W4394770144","doi":"10.1162/dint_a_00251","title":"LLaMA-LoRA Neural Prompt Engineering: A Deep Tuning Framework for Automatically Generating Chinese Text Logical Reasoning Thinking Chains","year":2024,"lang":"en","type":"article","venue":"Data Intelligence","topic":"Topic Modeling","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Language model; Artificial intelligence; Natural language processing; Comprehension; Benchmark (surveying); Inference; Logical reasoning; Question answering; Programming language","score_opus":0.05625293307651095,"score_gpt":0.3232568566701648,"score_spread":0.2670039235936538,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394770144","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041934106,0.0004320957,0.9411545,0.0006254954,0.000104235914,0.00019129328,0.00050737365,0.012183863,0.0028671413],"genre_scores_gemma":[0.58964795,0.00027666788,0.4003469,0.0006659911,0.00008256363,0.0005224294,0.0015402756,0.00039283754,0.006524399],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999645,0.00010588715,0.000023040406,0.00012276882,0.000059103422,0.000044286393],"domain_scores_gemma":[0.99908113,0.00042789945,0.00006243294,0.00010005664,0.00025152418,0.00007690626],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012673126,0.00085335626,0.0006103454,0.00067616755,0.00043698496,0.0010040447,0.0019565867,0.0011357474,0.0056091375],"category_scores_gemma":[0.0043233414,0.00045501717,0.0008995861,0.00047694234,0.00054305646,0.0017817894,0.0013696626,0.0023642406,0.0014501584],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031793953,0.00040264946,0.0028214839,0.0002496094,0.000097198375,0.00020759259,0.00030698662,0.3964435,0.012275206,0.01430817,0.010943519,0.56162614],"study_design_scores_gemma":[0.0000131040515,0.00003287789,0.00008988483,0.000007178894,0.000008090213,0.000009098275,0.000013125689,0.9935644,0.001065106,0.00444924,0.0007424815,0.0000053201625],"about_ca_topic_score_codex":0.008054056,"about_ca_topic_score_gemma":0.013599532,"teacher_disagreement_score":0.008054056,"about_ca_system_score_codex":0.0011374771,"about_ca_system_score_gemma":0.0017196464,"threshold_uncertainty_score":0.018764377},"labels":[],"label_agreement":null},{"id":"W4400522920","doi":"10.1162/dint_a_00255","title":"FAIR Enough: Develop and Assess a FAIR-Compliant Dataset for Large Language Model Training?","year":2024,"lang":"en","type":"article","venue":"Data Intelligence","topic":"Research Data Management Practices","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; Toronto Metropolitan University; Vector Institute","funders":"Government of Canada; Canadian Institute for Advanced Research","keywords":"Stewardship (theology); Computer science; Context (archaeology); Interoperability; Frame (networking); Process (computing); Work (physics); Engineering ethics; Knowledge management; Process management; Accreditation; Checklist; Data science; Risk analysis (engineering); Political science; Business; Engineering; Psychology; World Wide Web; Medical education; Medicine","score_opus":0.39620062148958934,"score_gpt":0.46542847373583285,"score_spread":0.0692278522462435,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400522920","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16354674,0.0007113074,0.7613403,0.018363515,0.0010795052,0.008098282,0.009038901,0.007946685,0.029874723],"genre_scores_gemma":[0.2352482,0.00017310404,0.7423065,0.0021815733,0.0000902631,0.0052400217,0.010945961,0.0014688936,0.0023455522],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.8604078,0.0983826,0.010138158,0.0072970614,0.02182547,0.0019488684],"domain_scores_gemma":[0.58396846,0.17912208,0.023202006,0.14127162,0.06554587,0.006889969],"candidate_categories":["metaresearch","open_science"],"consensus_categories":[],"category_scores_codex":[0.18429072,0.0009079298,0.0007970192,0.0043533845,0.0049234726,0.012495776,0.0056552035,0.0040053236,0.006338498],"category_scores_gemma":[0.43042952,0.0010363178,0.0013281697,0.003299801,0.006726117,0.01499865,0.016069802,0.0054969485,0.002269391],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018300908,0.0031717962,0.079650976,0.0026564912,0.0004449229,0.0010543041,0.04158372,0.03717316,0.012161486,0.28495222,0.108882464,0.4264384],"study_design_scores_gemma":[0.0011389861,0.0023121121,0.027534125,0.0069444273,0.0002865581,0.00097104296,0.026482156,0.17063579,0.05063511,0.340063,0.37223426,0.000762498],"about_ca_topic_score_codex":0.0058888155,"about_ca_topic_score_gemma":0.010711328,"teacher_disagreement_score":0.9943448,"about_ca_system_score_codex":0.005808251,"about_ca_system_score_gemma":0.01697844,"threshold_uncertainty_score":0.9746342},"labels":[],"label_agreement":null},{"id":"W4400831116","doi":"10.3724/2096-7004.di.2024.0002","title":"Risk Factors Categorizations of Ischemic Heart Disease in South-Western Bangladesh","year":2024,"lang":"en","type":"article","venue":"Data Intelligence","topic":"Artificial Intelligence in Healthcare","field":"Health Professions","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Disease; Voting; Population; Actuarial science; Medicine; Geography; Environmental health; Demography; Business; Political science; Pathology; Sociology","score_opus":0.21853472166579566,"score_gpt":0.4789780423696013,"score_spread":0.2604433207038056,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400831116","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9967006,0.00007331909,0.00030928987,0.00010405808,0.0000043888954,0.000038052396,0.0010102352,0.0000051523416,0.0017549524],"genre_scores_gemma":[0.9984042,0.00006455396,0.00034665104,0.000010379348,0.0000025312902,0.000026193931,0.0008709954,9.993346e-7,0.0002734762],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9997439,0.00006876106,0.000051430503,0.000043560256,0.00005351309,0.00003882985],"domain_scores_gemma":[0.99931955,0.0001945776,0.00015275223,0.0000553424,0.0001824735,0.000095368996],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00035316058,0.0002533179,0.00020092656,0.001600379,0.00034623512,0.00056109595,0.00020947005,0.00019176416,0.0019330574],"category_scores_gemma":[0.0019071456,0.00008203823,0.00026763638,0.0011708958,0.0002174733,0.00023886967,0.0004697145,0.00023735111,0.00041672838],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020049652,0.000055157,0.9794672,0.000033007036,0.000035228575,0.00016887815,0.0010085382,0.00033801104,0.0015240089,0.00027280088,0.0005040239,0.016392494],"study_design_scores_gemma":[0.000009665973,0.00009551389,0.9916373,0.000024542327,0.000024194254,0.00020276195,0.004611616,0.0017681948,0.0003593109,0.00035082435,0.0008997966,0.000016325237],"about_ca_topic_score_codex":0.014318302,"about_ca_topic_score_gemma":0.013044258,"teacher_disagreement_score":0.014318302,"about_ca_system_score_codex":0.00048609087,"about_ca_system_score_gemma":0.00034508604,"threshold_uncertainty_score":0.02846992},"labels":[],"label_agreement":null}]}