{"meta":{"query_hash":"cbcfb6b7cdf0","filters":{"venue":"NEJM AI"},"cohort_total":11,"direct_labels_cover":0,"predictions_cover":11,"exported":11,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/cbcfb6b7cdf0","api":"https://metacan.xera.ac/api/v1/cohort?venue=NEJM+AI"},"results":[{"id":"W4394845024","doi":"10.1056/aioa2300151","title":"Comparative Evaluation of LLMs in Clinical Oncology","year":2024,"lang":"en","type":"article","venue":"NEJM AI","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":97,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Sunnybrook Health Science Centre","funders":"National Cancer Institute; NIH Clinical Center; National Institutes of Health","keywords":"Clinical Oncology; Internal medicine; Oncology; Medicine; Cancer","score_opus":0.7465636406134145,"score_gpt":0.698507249034476,"score_spread":0.048056391578938484,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394845024","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.88207006,0.005648586,0.06475421,0.0034948264,0.00048170958,0.0008965242,0.006717852,0.023053551,0.012882627],"genre_scores_gemma":[0.9567631,0.00075827964,0.032431494,0.00046242453,0.00006852769,0.0003201512,0.00767375,0.0003284362,0.0011938137],"study_design_codex":"simulation_or_modeling","study_design_gemma":"observational","domain_scores_codex":[0.993604,0.0037888924,0.00060486683,0.0010601357,0.0007602062,0.00018185512],"domain_scores_gemma":[0.9526858,0.039489668,0.0013135012,0.0028056598,0.0027436076,0.0009617656],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012554849,0.0015936147,0.00074809324,0.002172422,0.0003997573,0.0018512813,0.002630424,0.0020352015,0.0037744325],"category_scores_gemma":[0.056435507,0.000497699,0.0013648656,0.0011902382,0.0007873148,0.0029631725,0.002304304,0.0021491076,0.0013911196],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0062115095,0.0025105062,0.07603617,0.0027198775,0.0014734983,0.00032603936,0.001269774,0.50990856,0.004945602,0.002977244,0.020664517,0.3709567],"study_design_scores_gemma":[0.0004409087,0.002541513,0.016049681,0.00030273662,0.00032559995,0.00031130837,0.00041812687,0.9592884,0.008375345,0.003914181,0.007926029,0.00010604407],"about_ca_topic_score_codex":0.0070988215,"about_ca_topic_score_gemma":0.0074198125,"teacher_disagreement_score":0.012554849,"about_ca_system_score_codex":0.0035068511,"about_ca_system_score_gemma":0.0017750654,"threshold_uncertainty_score":0.06639719},"labels":[],"label_agreement":null},{"id":"W4404759704","doi":"10.1056/aics2400639","title":"Cognitive Biases and Artificial Intelligence","year":2024,"lang":"en","type":"article","venue":"NEJM AI","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":35,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Health Sciences Centre; University of Toronto; Institute for Clinical Evaluative Sciences; Sunnybrook Health Science Centre","funders":"","keywords":"Cognition; Psychology; Cognitive psychology; Artificial intelligence; Cognitive science; Computer science; Neuroscience","score_opus":0.32906220684341303,"score_gpt":0.5014888764580899,"score_spread":0.1724266696146769,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404759704","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11143548,0.06236451,0.04087452,0.1882006,0.0032899126,0.00006813306,0.00029298422,0.0001552824,0.59331864],"genre_scores_gemma":[0.946216,0.015259182,0.009367559,0.008137932,0.0021624805,0.00010276297,0.000099359844,0.00006596429,0.018588834],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.997256,0.0015372374,0.00010925776,0.00023692606,0.00072593533,0.00013469625],"domain_scores_gemma":[0.97986436,0.016099278,0.0008145197,0.001275267,0.0014461324,0.0005004766],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0047092508,0.00037849424,0.0004400354,0.0021882525,0.0010252851,0.0057893433,0.0006059411,0.0024905838,0.005922658],"category_scores_gemma":[0.02746803,0.0002480316,0.00021682854,0.0016647468,0.0096653635,0.0060039023,0.0016009818,0.0023604296,0.00065043115],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000043695978,0.000054564895,0.0031507364,0.00016833014,0.000066888555,0.00008648339,0.0020259297,0.0014348768,0.00012836425,0.9309711,0.010652493,0.051216625],"study_design_scores_gemma":[0.000009160071,0.000005061934,0.00078435143,0.000059021266,0.000008706417,0.000043697048,0.00028279514,0.00058542244,0.000054103857,0.98501956,0.0131418165,0.000006299793],"about_ca_topic_score_codex":0.0018294373,"about_ca_topic_score_gemma":0.0017763571,"teacher_disagreement_score":0.005922658,"about_ca_system_score_codex":0.0015613172,"about_ca_system_score_gemma":0.0012750116,"threshold_uncertainty_score":0.024905145},"labels":[],"label_agreement":null},{"id":"W4405533437","doi":"10.1056/aip2401088","title":"Tackling Algorithmic Bias and Promoting Transparency in Health Datasets: The STANDING Together Consensus Recommendations","year":2024,"lang":"en","type":"article","venue":"NEJM AI","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Google (Canada); SickKids Foundation; Hospital for Sick Children","funders":"Economic and Social Research Council; Medical Research Council","keywords":"Transparency (behavior); Data science; Computer science; Internet privacy; Computer security","score_opus":0.22169389607060688,"score_gpt":0.47144389083435867,"score_spread":0.2497499947637518,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405533437","genre_codex":"commentary","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0021310044,0.06685043,0.09208041,0.8208026,0.011693139,0.00043012263,0.00088434893,0.00038030444,0.004747533],"genre_scores_gemma":[0.12659316,0.07860144,0.4230026,0.33332193,0.028998917,0.0024513365,0.002942228,0.0007061744,0.003382149],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.7968668,0.14499837,0.018605009,0.013612039,0.022892302,0.003025407],"domain_scores_gemma":[0.2728759,0.60709125,0.0144205345,0.04106692,0.05708606,0.007459299],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2874652,0.0021577126,0.0055746124,0.006158065,0.0033466711,0.016661579,0.010393188,0.020464454,0.010448492],"category_scores_gemma":[0.56499827,0.0017925227,0.0066747773,0.0052316994,0.013720346,0.029526567,0.011240145,0.03767472,0.0027957165],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010616241,0.0003403954,0.005164776,0.011945414,0.0049893768,0.00023128583,0.0018592494,0.0056242673,0.00042713576,0.31546402,0.24999648,0.40289587],"study_design_scores_gemma":[0.0006519587,0.00012704245,0.0010209515,0.013924863,0.0015640734,0.00013354259,0.00070263096,0.0071677715,0.0007633627,0.85146564,0.12230858,0.00016947501],"about_ca_topic_score_codex":0.0070323097,"about_ca_topic_score_gemma":0.010726983,"teacher_disagreement_score":0.7125348,"about_ca_system_score_codex":0.0054986225,"about_ca_system_score_gemma":0.025942877,"threshold_uncertainty_score":0.87868226},"labels":[],"label_agreement":null},{"id":"W4408023203","doi":"10.1056/aioa2400948","title":"Opportunistic Screening of Chronic Liver Disease with Deep-Learning–Enhanced Echocardiography","year":2025,"lang":"en","type":"article","venue":"NEJM AI","topic":"Liver Disease and Transplantation","field":"Medicine","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"National Center for Advancing Translational Sciences; National Heart, Lung, and Blood Institute","keywords":"Medicine; Cardiology; Internal medicine; Chronic liver disease; Cirrhosis","score_opus":0.007730231190715632,"score_gpt":0.24643949155183506,"score_spread":0.23870926036111942,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408023203","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.985201,0.0006580096,0.00996232,0.0004125779,0.000026628784,0.00014709804,0.0017771606,0.00032139552,0.00149382],"genre_scores_gemma":[0.9921787,0.00013150336,0.0061117252,0.00011522574,0.000024560342,0.00003906106,0.0012353932,0.000009330557,0.0001544674],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.998901,0.0005303865,0.000078523146,0.00026083455,0.00013355925,0.000095686824],"domain_scores_gemma":[0.9968635,0.0013179688,0.0007197293,0.0002846236,0.0004736437,0.0003405342],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020479127,0.00060498493,0.00044088,0.0010424883,0.00016253017,0.0005565844,0.00066707126,0.00038553454,0.0008472557],"category_scores_gemma":[0.006577026,0.0003019903,0.00047500458,0.00043229488,0.0002555051,0.0005826845,0.0009793446,0.00057672715,0.00030399562],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008095126,0.00041287977,0.941024,0.00007658558,0.00022194083,0.0001784367,0.000049703613,0.010766834,0.0011737524,0.000111017886,0.0015436675,0.04363167],"study_design_scores_gemma":[0.00021006225,0.0014960704,0.46360862,0.00012669775,0.00027500276,0.0011223061,0.00012989939,0.5248906,0.0045182006,0.0011951295,0.0023575232,0.00006990785],"about_ca_topic_score_codex":0.0046130675,"about_ca_topic_score_gemma":0.010848664,"teacher_disagreement_score":0.0046130675,"about_ca_system_score_codex":0.0005619922,"about_ca_system_score_gemma":0.00064276357,"threshold_uncertainty_score":0.010830522},"labels":[],"label_agreement":null},{"id":"W4409767654","doi":"10.1056/aioa2400703","title":"Longitudinal Risk Prediction for Pediatric Glioma with Temporal Deep Learning","year":2025,"lang":"en","type":"article","venue":"NEJM AI","topic":"Glioma Diagnosis and Treatment","field":"Medicine","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"National Institute of Dental and Craniofacial Research; National Cancer Institute","keywords":"Deep learning; Glioma; Artificial intelligence; Computer science; Medicine; Cancer research","score_opus":0.009416397693304265,"score_gpt":0.2674186025855746,"score_spread":0.2580022048922703,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409767654","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.87343436,0.0021656468,0.11426674,0.001821469,0.00010002462,0.00007711456,0.0047987807,0.0014549992,0.0018807132],"genre_scores_gemma":[0.974878,0.0003994175,0.01918459,0.00019681094,0.000049913928,0.000056571596,0.004322798,0.000036875263,0.0008749368],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99974877,0.000065229695,0.000021193891,0.00008542972,0.00004046606,0.000038822636],"domain_scores_gemma":[0.99916327,0.00032538414,0.00020597583,0.0000702274,0.00016505792,0.0000700762],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010666553,0.0006812753,0.0003794963,0.0008029588,0.00017291268,0.0004211415,0.00060793076,0.0004175834,0.0008487393],"category_scores_gemma":[0.003068855,0.00024009787,0.0006976604,0.00048286747,0.00018755817,0.0005251917,0.000633804,0.0008902412,0.0002734741],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010294907,0.00054863095,0.3595159,0.00014181137,0.0004410089,0.000740077,0.00011834136,0.3814404,0.003635476,0.0015925673,0.011208715,0.23958766],"study_design_scores_gemma":[0.00002543333,0.00011081226,0.010823901,0.00002075499,0.000054760552,0.00016980513,0.000024792118,0.9844332,0.0013935311,0.0021711916,0.0007572528,0.000014713197],"about_ca_topic_score_codex":0.013375106,"about_ca_topic_score_gemma":0.018425046,"teacher_disagreement_score":0.013375106,"about_ca_system_score_codex":0.0008989636,"about_ca_system_score_gemma":0.0011657597,"threshold_uncertainty_score":0.02659452},"labels":[],"label_agreement":null},{"id":"W4409990539","doi":"10.1056/aics2401008","title":"Clinical Deployment of Real-Time Left Ventricular Ejection Fraction Estimation from Coronary Angiography","year":2025,"lang":"en","type":"article","venue":"NEJM AI","topic":"Cardiac Imaging and Diagnostics","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary; Université de Montréal; Montreal Heart Institute; University of Ottawa","funders":"","keywords":"Ejection fraction; Cardiology; Coronary angiography; Internal medicine; Software deployment; Angiography; Fraction (chemistry); Estimation; Medicine; Computer science; Heart failure; Myocardial infarction; Economics; Chemistry","score_opus":0.010107056097290344,"score_gpt":0.3206113315904624,"score_spread":0.31050427549317206,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409990539","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.71152484,0.016194817,0.22237575,0.00723746,0.0017472152,0.00075412134,0.0012116071,0.004338225,0.034616075],"genre_scores_gemma":[0.95533395,0.0021558746,0.037848342,0.0014639192,0.00093523524,0.00014515263,0.00033820656,0.00018495588,0.0015944354],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99901354,0.00047985258,0.00007030533,0.00016987276,0.00019915386,0.000067342466],"domain_scores_gemma":[0.9975904,0.0012425007,0.00018454452,0.00024153532,0.00047233305,0.00026863618],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018339324,0.0006009841,0.0006268346,0.0006807395,0.00021047668,0.001590879,0.00061256625,0.0011988806,0.0033363488],"category_scores_gemma":[0.007421003,0.00039534125,0.0001989271,0.0002529754,0.0002948722,0.0007572096,0.0006120457,0.0008663978,0.0018893111],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.010800865,0.0010371875,0.14822106,0.0006004189,0.0001845077,0.0072076903,0.00051456416,0.0040077968,0.13418347,0.0015923794,0.01706825,0.67458177],"study_design_scores_gemma":[0.0025176154,0.014746081,0.44865435,0.0012765483,0.0011470164,0.048474994,0.00141594,0.23483665,0.15841657,0.0074137403,0.08051253,0.0005879888],"about_ca_topic_score_codex":0.0005492768,"about_ca_topic_score_gemma":0.0009039871,"teacher_disagreement_score":0.0033363488,"about_ca_system_score_codex":0.00016641681,"about_ca_system_score_gemma":0.00035981898,"threshold_uncertainty_score":0.011161149},"labels":[],"label_agreement":null},{"id":"W4411415068","doi":"10.1056/aipc2500153","title":"Lessons from the Failure of Canada’s Artificial Intelligence and Data Act","year":2025,"lang":"en","type":"article","venue":"NEJM AI","topic":"Ethics in Clinical Research","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Princess Margaret Cancer Centre; University Health Network; University of Toronto","funders":"","keywords":"Artificial intelligence; Computer science","score_opus":0.5619306336519689,"score_gpt":0.5951833525381953,"score_spread":0.033252718886226384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411415068","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018730265,0.0021800527,0.00055559305,0.9784782,0.0010566374,0.000009955692,0.00007516617,0.000019253579,0.015752103],"genre_scores_gemma":[0.1716686,0.0038238645,0.0018967683,0.8003349,0.0034192712,0.00008301396,0.00011697514,0.00018055199,0.01847618],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.93763494,0.020934274,0.0020565505,0.0040818183,0.021778932,0.013513425],"domain_scores_gemma":[0.7311477,0.14674567,0.006358236,0.007769039,0.059253264,0.048726164],"candidate_categories":["sts"],"consensus_categories":[],"category_scores_codex":[0.0740294,0.00082675536,0.001974508,0.0028778075,0.02551053,0.024503559,0.005742964,0.046609487,0.009878671],"category_scores_gemma":[0.16443446,0.0011704938,0.0014067823,0.0033331001,0.0551257,0.009665139,0.007953755,0.06194771,0.0014068378],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000114593706,0.000059684655,0.0037244915,0.00009475767,0.00009526981,0.00064204546,0.0051204516,0.0011519862,0.000113310954,0.5147103,0.45443735,0.0197356],"study_design_scores_gemma":[0.00028476748,0.00005212126,0.006530741,0.0012098826,0.00009995354,0.0003322554,0.006061274,0.001425881,0.00032094793,0.29252878,0.6908819,0.00027142503],"about_ca_topic_score_codex":0.9542326,"about_ca_topic_score_gemma":0.95958614,"teacher_disagreement_score":0.97448945,"about_ca_system_score_codex":0.13662042,"about_ca_system_score_gemma":0.33744264,"threshold_uncertainty_score":0.99125516},"labels":[],"label_agreement":null},{"id":"W4411672381","doi":"10.1056/aioa2401221","title":"Expert-Level Detection of Epilepsy Markers in EEG on Short and Long Timescales","year":2025,"lang":"en","type":"article","venue":"NEJM AI","topic":"EEG and Brain-Computer Interfaces","field":"Neuroscience","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"National Center for Advancing Translational Sciences; National Institute of Neurological Disorders and Stroke; National Institute of General Medical Sciences; National Heart, Lung, and Blood Institute; National Institute on Aging; U.S. Department of Veterans Affairs","keywords":"Epilepsy; Electroencephalography; Psychology; Audiology; Neuroscience; Pattern recognition (psychology); Computer science; Medicine; Cognitive psychology","score_opus":0.026709074447145105,"score_gpt":0.2920409229830009,"score_spread":0.2653318485358558,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411672381","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.888405,0.001267064,0.102782145,0.00060465763,0.00006529387,0.000119291464,0.0024091164,0.001020922,0.0033263674],"genre_scores_gemma":[0.9852844,0.00017626189,0.01233399,0.00008072713,0.000034293687,0.00002433782,0.0014331987,0.00001828674,0.0006145219],"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9992084,0.0002417716,0.00007360086,0.0002589755,0.00015266809,0.0000645969],"domain_scores_gemma":[0.99671435,0.0016978927,0.00053512736,0.00026814445,0.0006563553,0.00012817557],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002186498,0.00059488235,0.0003577455,0.0010948881,0.00011329064,0.00071245077,0.00042252778,0.00056066184,0.0010788995],"category_scores_gemma":[0.0095778005,0.000115812174,0.00031187903,0.0005567476,0.00019729072,0.00064152613,0.00056184625,0.0005479805,0.00047753283],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00129348,0.00041870802,0.48177454,0.00038827112,0.0005226678,0.0005624369,0.00038857528,0.108437926,0.01926387,0.0008358746,0.008184797,0.37792894],"study_design_scores_gemma":[0.00006657916,0.00044661746,0.23688824,0.00010004501,0.00015773051,0.000957469,0.000226396,0.7369502,0.017981708,0.0027670558,0.0034043663,0.000053549105],"about_ca_topic_score_codex":0.0034502626,"about_ca_topic_score_gemma":0.0064682993,"teacher_disagreement_score":0.0034502626,"about_ca_system_score_codex":0.0003713642,"about_ca_system_score_gemma":0.0004829142,"threshold_uncertainty_score":0.01156342},"labels":[],"label_agreement":null},{"id":"W4413815394","doi":"10.1056/aip2500154","title":"Energy Considerations for Scaling Artificial Intelligence Adoption in Medicine: First Do No Harm","year":2025,"lang":"en","type":"article","venue":"NEJM AI","topic":"Climate Change and Health Impacts","field":"Environmental Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Public Health; Public Health Ontario; University of Toronto","funders":"","keywords":"Harm; Do no harm; Scaling; Energy (signal processing); Computer science; Artificial intelligence; Psychology; Social psychology; Mathematics; Psychiatry; Statistics","score_opus":0.10835591873721445,"score_gpt":0.3702471005966559,"score_spread":0.26189118185944144,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413815394","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012355412,0.025857735,0.06573028,0.6880307,0.011351184,0.00026066127,0.0005011839,0.0005422063,0.19537061],"genre_scores_gemma":[0.67846835,0.024598988,0.08702333,0.16073065,0.011849331,0.0008849556,0.00039529268,0.0008583855,0.03519066],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9885439,0.005620471,0.00049711886,0.00072129606,0.0037100313,0.0009071288],"domain_scores_gemma":[0.95799845,0.024358518,0.0016308062,0.004465042,0.009258764,0.002288514],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014913157,0.0010716425,0.0011562543,0.001941233,0.0025828346,0.008733656,0.0025846446,0.00949845,0.026097672],"category_scores_gemma":[0.07807539,0.00061501126,0.0016016816,0.0012244537,0.006897179,0.015509224,0.0063631153,0.009071028,0.00449016],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046018334,0.00034651594,0.0027469096,0.0010141699,0.0003742745,0.0002956032,0.0005897448,0.013884518,0.0027667936,0.6545333,0.11725414,0.20573373],"study_design_scores_gemma":[0.00014911675,0.0004894697,0.0035720076,0.0020972644,0.00013599459,0.0004484256,0.001614891,0.0073420173,0.0017275589,0.67369294,0.30855492,0.00017537351],"about_ca_topic_score_codex":0.005081927,"about_ca_topic_score_gemma":0.0063155256,"teacher_disagreement_score":0.026097672,"about_ca_system_score_codex":0.003435721,"about_ca_system_score_gemma":0.008553125,"threshold_uncertainty_score":0.08730543},"labels":[],"label_agreement":null},{"id":"W4414499338","doi":"10.1056/aidbp2500120","title":"Assessment of Large Language Models in Clinical Reasoning: A Novel Benchmarking Study","year":2025,"lang":"en","type":"article","venue":"NEJM AI","topic":"Clinical Reasoning and Diagnostic Skills","field":"Medicine","cited_by":28,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Benchmarking; Language model; Quality (philosophy); Identification (biology); Measure (data warehouse)","score_opus":0.037095345024797416,"score_gpt":0.45895407758977547,"score_spread":0.4218587325649781,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414499338","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8368357,0.0016646661,0.1446742,0.0013378531,0.00017391206,0.00088420283,0.002644619,0.0017175933,0.0100672925],"genre_scores_gemma":[0.93934804,0.0002142243,0.054900747,0.0002796171,0.000048039044,0.00039063292,0.0036733232,0.00025787353,0.000887538],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9691153,0.02262321,0.0020620767,0.0025341648,0.0031608862,0.0005044734],"domain_scores_gemma":[0.79654765,0.16631934,0.0046861684,0.017272728,0.012998377,0.0021757507],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.032850422,0.0011761793,0.000924357,0.002170961,0.0010064596,0.0038319265,0.002854826,0.0021152755,0.0041765957],"category_scores_gemma":[0.16919091,0.0004690127,0.0013140417,0.0020203143,0.0013101939,0.0065182424,0.004152421,0.0022807426,0.00091697613],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.01109319,0.010512992,0.16107607,0.0021976249,0.0020700234,0.00089739606,0.0065238876,0.23704645,0.009432373,0.023933372,0.01963533,0.5155813],"study_design_scores_gemma":[0.0013318062,0.0045838137,0.036957007,0.00041347515,0.0007142548,0.00073980977,0.0027054315,0.8999142,0.007938802,0.034201518,0.010326126,0.00017373811],"about_ca_topic_score_codex":0.0055613373,"about_ca_topic_score_gemma":0.005595571,"teacher_disagreement_score":0.032850422,"about_ca_system_score_codex":0.002375114,"about_ca_system_score_gemma":0.0025067434,"threshold_uncertainty_score":0.17373174},"labels":[],"label_agreement":null},{"id":"W7117104805","doi":"10.1056/aioa2500522","title":"International Retrospective Observational Study of Continual Learning for AI on Endotracheal Tube Placement from Chest Radiographs","year":2025,"lang":"en","type":"article","venue":"NEJM AI","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Health Sciences Centre; Sunnybrook Health Science Centre; Western University; University of Toronto","funders":"","keywords":"Observational study; Radiography; Retrospective cohort study; Endotracheal tube; MEDLINE","score_opus":0.1655640285438163,"score_gpt":0.4549574241101151,"score_spread":0.2893933955662988,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7117104805","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9984207,0.00032108626,0.00008119919,0.000038734037,0.000010066951,0.000029639476,0.00041111052,0.0000031128702,0.00068435975],"genre_scores_gemma":[0.9985643,0.00021472583,0.00012285112,0.00010653844,0.000035827594,0.000034298548,0.00072939036,0.0000051856005,0.00018688897],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9986738,0.00027936013,0.00028321194,0.0003451753,0.00020797439,0.00021052506],"domain_scores_gemma":[0.9930917,0.0010047259,0.003717089,0.00054737256,0.00067610247,0.00096303865],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00094717863,0.00047180787,0.0006454238,0.0012998626,0.0008798305,0.0011327966,0.0007181988,0.001049667,0.00254839],"category_scores_gemma":[0.0048080524,0.00061722356,0.00074844295,0.0021747157,0.0006673161,0.00093602866,0.0008788058,0.0013912765,0.0006363146],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010707319,0.00010443277,0.9986842,0.000020726602,0.000041614236,0.00022733855,0.00012901568,0.000007364322,0.00005624411,0.000013834467,0.000099928424,0.000508232],"study_design_scores_gemma":[0.000016225098,0.0004178767,0.99608094,0.000034974055,0.000063441126,0.001577998,0.001283227,0.000071148126,0.000040691517,0.000022574897,0.00037856225,0.000012290604],"about_ca_topic_score_codex":0.0059981886,"about_ca_topic_score_gemma":0.005307574,"teacher_disagreement_score":0.0059981886,"about_ca_system_score_codex":0.0006423729,"about_ca_system_score_gemma":0.0011038493,"threshold_uncertainty_score":0.011926591},"labels":[],"label_agreement":null}]}