{"meta":{"query_hash":"cbcfb6b7cdf0","filters":{"venue":"NEJM AI"},"cohort_total":11,"direct_labels_cover":0,"predictions_cover":11,"exported":11,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/cbcfb6b7cdf0","api":"https://metacan.xera.ac/api/v1/cohort?venue=NEJM+AI"},"results":[{"id":"W4394845024","doi":"10.1056/aioa2300151","title":"Comparative Evaluation of LLMs in Clinical Oncology","year":2024,"lang":"en","type":"article","venue":"NEJM AI","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":97,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Sunnybrook Health Science Centre","funders":"National Cancer Institute; NIH Clinical Center; National Institutes of Health","keywords":"Clinical Oncology; Internal medicine; Oncology; Medicine; Cancer","score_opus":0.7465636406134145,"score_gpt":0.698507249034476,"score_spread":0.048056391578938484,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394845024","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9815709,0.0008751158,0.00007900089,0.00673067,0.0011151736,0.0002977347,8.11407e-7,0.000014234651,0.009316374],"genre_scores_gemma":[0.99882585,0.00005495475,0.00011823366,0.000527098,0.0003142745,0.000022619895,0.000011033436,0.0000032927403,0.00012264731],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9989985,0.00017286239,0.00044927973,0.000121785655,0.00017913427,0.00007840811],"domain_scores_gemma":[0.9992801,0.0003125084,0.00003612934,0.000091223344,0.00023692472,0.000043139353],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017287885,0.000036001537,0.00018798857,0.00008538419,0.000009835803,0.0000031231107,0.00002274613,0.00009745132,0.00029885845],"category_scores_gemma":[0.00029962306,0.000030250083,0.000039567694,0.00019061322,0.00007225616,0.0000435092,0.000005623258,0.00021346471,0.00016845381],"study_design_candidate":"design_other","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014715384,0.0003893924,0.03655686,0.0001066305,0.000028494656,0.000007364754,0.013668313,0.00008336497,0.0005492608,0.0017755884,0.011217884,0.9354697],"study_design_scores_gemma":[0.00068010035,0.005586064,0.43019888,0.0015011871,0.00058886374,0.000050578496,0.025346784,0.34918168,0.042113367,0.03956284,0.10487358,0.00031608387],"about_ca_topic_score_codex":0.00038389134,"about_ca_topic_score_gemma":0.00038789853,"teacher_disagreement_score":0.9351536,"about_ca_system_score_codex":0.00014693999,"about_ca_system_score_gemma":0.0010720469,"threshold_uncertainty_score":0.32722902},"labels":[],"label_agreement":null},{"id":"W4404759704","doi":"10.1056/aics2400639","title":"Cognitive Biases and Artificial Intelligence","year":2024,"lang":"en","type":"article","venue":"NEJM AI","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":35,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Health Sciences Centre; University of Toronto; Institute for Clinical Evaluative Sciences; Sunnybrook Health Science Centre","funders":"","keywords":"Cognition; Psychology; Cognitive psychology; Artificial intelligence; Cognitive science; Computer science; Neuroscience","score_opus":0.32906220684341303,"score_gpt":0.5014888764580899,"score_spread":0.1724266696146769,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404759704","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9699206,0.003126215,0.005579621,0.017227925,0.0011178136,0.00031790845,0.00000912008,0.00016647606,0.002534366],"genre_scores_gemma":[0.99679744,0.00026670654,0.00011402002,0.0017136345,0.00069344434,0.000019765686,0.000019473116,0.000012173055,0.00036335312],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99929154,0.000018178229,0.00020625812,0.00021502843,0.00011251559,0.00015645703],"domain_scores_gemma":[0.9992283,0.00046405467,0.000012929274,0.00007755286,0.00010479635,0.0001123449],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00013858195,0.00007687574,0.00010261256,0.000105283136,0.000069822374,0.000055824195,0.00002148325,0.00006734056,0.00036922356],"category_scores_gemma":[0.00058398646,0.000065959,0.000032296706,0.00022241438,0.00012918061,0.00009660448,0.000013065263,0.00018760024,0.0004937021],"study_design_candidate":"design_other","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000062945095,0.000050584866,0.0009554294,0.00010632421,0.000017699882,0.000045249948,0.0032240108,0.0000012516697,0.0005707649,0.008012432,0.0014316512,0.9855217],"study_design_scores_gemma":[0.000054099535,0.002750261,0.0063858936,0.008538535,0.0006434484,0.0008706357,0.052707154,0.047326382,0.622517,0.21571448,0.041487057,0.0010050748],"about_ca_topic_score_codex":0.00022837058,"about_ca_topic_score_gemma":0.000049054055,"teacher_disagreement_score":0.98451656,"about_ca_system_score_codex":0.000028065466,"about_ca_system_score_gemma":0.00016512566,"threshold_uncertainty_score":0.6345706},"labels":[],"label_agreement":null},{"id":"W4405533437","doi":"10.1056/aip2401088","title":"Tackling Algorithmic Bias and Promoting Transparency in Health Datasets: The STANDING Together Consensus Recommendations","year":2024,"lang":"en","type":"article","venue":"NEJM AI","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Google (Canada); SickKids Foundation; Hospital for Sick Children","funders":"Economic and Social Research Council; Medical Research Council","keywords":"Transparency (behavior); Data science; Computer science; Internet privacy; Computer security","score_opus":0.22169389607060688,"score_gpt":0.47144389083435867,"score_spread":0.2497499947637518,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405533437","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.57928705,0.008298136,0.0026712765,0.40651718,0.001305537,0.0012668412,0.00019137081,0.00013483292,0.00032778695],"genre_scores_gemma":[0.9961328,0.00075920345,0.0008167501,0.0017851145,0.00023895709,0.00003053302,0.00014511321,0.000014796098,0.00007672278],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9989799,0.00010592953,0.0004037502,0.00020341072,0.00009775358,0.00020923388],"domain_scores_gemma":[0.9993702,0.00034197164,0.00004081117,0.0001419293,0.000027296752,0.00007778591],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010194347,0.000076210075,0.00012277291,0.00010268243,0.0001682206,0.000050010924,0.000037035534,0.000043095904,0.00009328382],"category_scores_gemma":[0.00017036356,0.000055594544,0.000019286892,0.0002816285,0.000055191613,0.00006489945,0.000008965188,0.00031522746,0.000015746946],"study_design_candidate":"design_other","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000026915117,0.00007630953,0.009402825,0.0008988849,0.000025492192,0.000023227309,0.05057096,0.000017632812,0.00027898926,0.0013551465,0.011404812,0.9259188],"study_design_scores_gemma":[0.0005361788,0.0014603244,0.0060146167,0.016027063,0.00017904636,0.000649288,0.10687223,0.18670729,0.007060692,0.019386806,0.654234,0.0008724767],"about_ca_topic_score_codex":0.0018068486,"about_ca_topic_score_gemma":0.00068310875,"teacher_disagreement_score":0.9250463,"about_ca_system_score_codex":0.000121469166,"about_ca_system_score_gemma":0.000270566,"threshold_uncertainty_score":0.27314267},"labels":[],"label_agreement":null},{"id":"W4408023203","doi":"10.1056/aioa2400948","title":"Opportunistic Screening of Chronic Liver Disease with Deep-Learning–Enhanced Echocardiography","year":2025,"lang":"en","type":"article","venue":"NEJM AI","topic":"Liver Disease and Transplantation","field":"Medicine","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"National Center for Advancing Translational Sciences; National Heart, Lung, and Blood Institute","keywords":"Medicine; Cardiology; Internal medicine; Chronic liver disease; Cirrhosis","score_opus":0.007730231190715632,"score_gpt":0.24643949155183506,"score_spread":0.23870926036111942,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408023203","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.72497994,0.0037497522,0.25832832,0.00049370964,0.00012911548,0.000862244,0.000045921824,0.0002170196,0.011194],"genre_scores_gemma":[0.99839956,0.00046699564,0.0002467087,0.0003037596,0.00005103,0.000019951403,0.00014372659,0.000011576324,0.0003566772],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99932206,0.000030910895,0.0001297719,0.00018607188,0.00018918882,0.00014201271],"domain_scores_gemma":[0.99949694,0.000035204936,0.000044649878,0.00017569118,0.00010026163,0.00014726515],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000046822213,0.000101955135,0.00018142353,0.00016761723,0.00005705451,0.000007825936,0.00004714431,0.00003317826,0.00008226492],"category_scores_gemma":[0.0000065554686,0.0000851693,0.00011040431,0.00030054522,0.0000892697,0.00007055816,0.000009269402,0.00012707799,0.00000765443],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.008647961,0.000725623,0.8669515,0.006313301,0.0045803837,0.0014496944,0.0012846731,0.0020398565,0.007846926,0.004704109,0.0005565872,0.094899386],"study_design_scores_gemma":[0.004038856,0.00056490174,0.97978294,0.0016452738,0.004582075,0.0000046468617,0.00014270109,0.0044355923,0.0037129172,0.00012950404,0.00076064013,0.00019997764],"about_ca_topic_score_codex":0.000032131862,"about_ca_topic_score_gemma":0.000006535408,"teacher_disagreement_score":0.27341965,"about_ca_system_score_codex":0.000020132597,"about_ca_system_score_gemma":0.00019004938,"threshold_uncertainty_score":0.34731033},"labels":[],"label_agreement":null},{"id":"W4409767654","doi":"10.1056/aioa2400703","title":"Longitudinal Risk Prediction for Pediatric Glioma with Temporal Deep Learning","year":2025,"lang":"en","type":"article","venue":"NEJM AI","topic":"Glioma Diagnosis and Treatment","field":"Medicine","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"National Institute of Dental and Craniofacial Research; National Cancer Institute","keywords":"Deep learning; Glioma; Artificial intelligence; Computer science; Medicine; Cancer research","score_opus":0.009416397693304265,"score_gpt":0.2674186025855746,"score_spread":0.2580022048922703,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409767654","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9929849,0.0008657656,0.0036865263,0.0007690382,0.00015932748,0.0005356564,0.000012013967,0.00010538548,0.00088141765],"genre_scores_gemma":[0.99798113,0.00007661997,0.0010631775,0.000078720805,0.0002566535,0.00017837118,0.00006002826,0.000014348272,0.00029093272],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9992722,0.000018605331,0.00014844244,0.00023779654,0.0001400701,0.00018287165],"domain_scores_gemma":[0.9995473,0.00007322921,0.0000656569,0.00013910835,0.00010622487,0.00006847593],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00009279453,0.00011692407,0.00018102933,0.00019050273,0.00015665677,0.000021685331,0.000031064057,0.000063219115,0.000030484753],"category_scores_gemma":[0.0000691789,0.0000850523,0.00007706255,0.0003373496,0.000022823908,0.000055257136,0.000014390729,0.00013606607,0.000014972334],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028627284,0.00023993183,0.9902227,0.00007172833,0.00012052861,0.000057421435,0.000060092963,0.00012739793,0.00004541194,0.0001297224,0.0026696464,0.0059691337],"study_design_scores_gemma":[0.005839469,0.0016007316,0.9790975,0.00006617406,0.0011921811,0.000046329253,0.000101729325,0.002220284,0.0011926782,0.00012898806,0.008416591,0.00009733545],"about_ca_topic_score_codex":0.00008235294,"about_ca_topic_score_gemma":0.00004337747,"teacher_disagreement_score":0.011125203,"about_ca_system_score_codex":0.00006650409,"about_ca_system_score_gemma":0.000059381586,"threshold_uncertainty_score":0.3468332},"labels":[],"label_agreement":null},{"id":"W4409990539","doi":"10.1056/aics2401008","title":"Clinical Deployment of Real-Time Left Ventricular Ejection Fraction Estimation from Coronary Angiography","year":2025,"lang":"en","type":"article","venue":"NEJM AI","topic":"Cardiac Imaging and Diagnostics","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary; Université de Montréal; Montreal Heart Institute; University of Ottawa","funders":"","keywords":"Ejection fraction; Cardiology; Coronary angiography; Internal medicine; Software deployment; Angiography; Fraction (chemistry); Estimation; Medicine; Computer science; Heart failure; Myocardial infarction; Economics; Chemistry","score_opus":0.010107056097290344,"score_gpt":0.3206113315904624,"score_spread":0.31050427549317206,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409990539","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9763944,0.0009891363,0.016415672,0.0010334023,0.0009424278,0.00033472496,0.000020952302,0.00015710219,0.0037121617],"genre_scores_gemma":[0.99634683,0.0001740893,0.002469889,0.00049821264,0.00016082964,0.0000062409904,0.00020731428,0.000010231221,0.0001263473],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9991802,0.000059888098,0.00030546275,0.00018168845,0.00017738067,0.000095363655],"domain_scores_gemma":[0.9992277,0.00029886924,0.00008739164,0.00022489272,0.000109263696,0.00005187278],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00017641325,0.00008118356,0.00024907853,0.00016661074,0.000035824618,0.00000815073,0.000026105012,0.000098020864,0.000057292153],"category_scores_gemma":[0.00022998058,0.00007992513,0.00031702933,0.00020747687,0.00004915727,0.00005840578,0.00001822933,0.00013250555,0.000042232783],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015450294,0.000290153,0.9559777,0.000029318806,0.00022289323,0.000028664013,0.000019541501,0.00017471571,0.0009425152,0.000027898157,0.022033654,0.020098468],"study_design_scores_gemma":[0.0011404373,0.00014285758,0.9875143,0.00020197054,0.00068647205,0.000008759832,0.000031452335,0.0046248366,0.0019469226,0.00040729126,0.0032290209,0.00006564015],"about_ca_topic_score_codex":0.00031198043,"about_ca_topic_score_gemma":0.0000016346381,"teacher_disagreement_score":0.031536665,"about_ca_system_score_codex":0.000052684325,"about_ca_system_score_gemma":0.000061646075,"threshold_uncertainty_score":0.3259252},"labels":[],"label_agreement":null},{"id":"W4411415068","doi":"10.1056/aipc2500153","title":"Lessons from the Failure of Canada’s Artificial Intelligence and Data Act","year":2025,"lang":"en","type":"article","venue":"NEJM AI","topic":"Ethics in Clinical Research","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Princess Margaret Cancer Centre; University Health Network; University of Toronto","funders":"","keywords":"Artificial intelligence; Computer science","score_opus":0.5619306336519689,"score_gpt":0.5951833525381953,"score_spread":0.033252718886226384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411415068","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08411624,0.0006325365,0.0019156663,0.90436924,0.0003025908,0.00029091357,0.0003107973,0.000013673281,0.008048314],"genre_scores_gemma":[0.9944876,0.00015180378,0.00040887872,0.003742829,0.000103277845,0.0000017233147,0.000028207942,0.000004344418,0.0010713239],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99895924,0.00005299428,0.00021553917,0.00023357033,0.00040628653,0.0001323792],"domain_scores_gemma":[0.99164915,0.0071567567,0.000029748095,0.0009489881,0.00015003058,0.000065323045],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0011283521,0.00004992571,0.0001414023,0.00001679697,0.00006923597,0.000014837672,0.00045840265,0.00011321861,0.0001322764],"category_scores_gemma":[0.013358042,0.000032523025,0.000014424651,0.00012765692,0.00029668913,0.000025792871,0.00045417907,0.0009771114,0.0000029424916],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043072007,0.00020941294,0.031401392,0.0003337456,0.00033261764,0.00006669385,0.0010234953,0.000010444301,0.0049420437,0.43337005,0.32826567,0.19961374],"study_design_scores_gemma":[0.0004979846,0.00021705632,0.10267001,0.0016146906,0.0002792098,0.00000391062,0.005419393,0.00523716,0.03336379,0.6170217,0.23342326,0.00025184185],"about_ca_topic_score_codex":0.22940648,"about_ca_topic_score_gemma":0.90592486,"teacher_disagreement_score":0.91037136,"about_ca_system_score_codex":0.000028667668,"about_ca_system_score_gemma":0.0032485935,"threshold_uncertainty_score":0.99495286},"labels":[],"label_agreement":null},{"id":"W4411672381","doi":"10.1056/aioa2401221","title":"Expert-Level Detection of Epilepsy Markers in EEG on Short and Long Timescales","year":2025,"lang":"en","type":"article","venue":"NEJM AI","topic":"EEG and Brain-Computer Interfaces","field":"Neuroscience","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"National Center for Advancing Translational Sciences; National Institute of Neurological Disorders and Stroke; National Institute of General Medical Sciences; National Heart, Lung, and Blood Institute; National Institute on Aging; U.S. Department of Veterans Affairs","keywords":"Epilepsy; Electroencephalography; Psychology; Audiology; Neuroscience; Pattern recognition (psychology); Computer science; Medicine; Cognitive psychology","score_opus":0.026709074447145105,"score_gpt":0.2920409229830009,"score_spread":0.2653318485358558,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411672381","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9957747,0.000080451195,0.0010949261,0.00040035733,0.00025721238,0.00010004772,0.0000035607432,0.000024986624,0.0022637565],"genre_scores_gemma":[0.99802727,0.000019366444,0.000048485992,0.0014259352,0.000013717269,0.000006413167,2.7248984e-7,0.0000045673037,0.0004539643],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.999354,0.000059242913,0.00014091906,0.00023616727,0.00009342889,0.00011623271],"domain_scores_gemma":[0.9996512,0.00017412906,0.000019025143,0.00012366488,0.00001030195,0.000021691376],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00008739921,0.000077230856,0.000108179,0.00013240158,0.00003365758,0.000020633966,0.000106169144,0.00004814644,0.000013247498],"category_scores_gemma":[0.00007198676,0.00006828967,0.000024644643,0.00016510455,0.00008462627,0.00007681171,0.00005017696,0.000106772066,0.000004003718],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014114946,0.00008951428,0.0074971193,0.000037389294,0.0000047445956,0.000012496982,0.0004051502,0.00009196572,0.85259074,0.00025667294,0.0012591984,0.13761388],"study_design_scores_gemma":[0.00022441313,0.00012296505,0.07918168,0.00013429431,0.000002103569,0.000004744311,0.000049552746,0.004827705,0.9139595,0.00015799468,0.0012530189,0.000082008104],"about_ca_topic_score_codex":0.000033867494,"about_ca_topic_score_gemma":0.00007549927,"teacher_disagreement_score":0.13753188,"about_ca_system_score_codex":0.00001889725,"about_ca_system_score_gemma":0.000011461905,"threshold_uncertainty_score":0.2784772},"labels":[],"label_agreement":null},{"id":"W4413815394","doi":"10.1056/aip2500154","title":"Energy Considerations for Scaling Artificial Intelligence Adoption in Medicine: First Do No Harm","year":2025,"lang":"en","type":"article","venue":"NEJM AI","topic":"Climate Change and Health Impacts","field":"Environmental Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Public Health; Public Health Ontario; University of Toronto","funders":"","keywords":"Harm; Do no harm; Scaling; Energy (signal processing); Computer science; Artificial intelligence; Psychology; Social psychology; Mathematics; Psychiatry; Statistics","score_opus":0.10835591873721445,"score_gpt":0.3702471005966559,"score_spread":0.26189118185944144,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413815394","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37522948,0.0012552147,0.17829998,0.32049653,0.006435978,0.0030912585,0.000120053366,0.00033506093,0.11473644],"genre_scores_gemma":[0.99160856,0.00017126407,0.0007451289,0.007048741,0.00016039504,0.0000595425,0.000011469519,0.000005529174,0.0001893493],"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9992544,0.000015168041,0.00024170155,0.00017881658,0.00009570573,0.00021422598],"domain_scores_gemma":[0.9995443,0.00023314578,0.00003572825,0.000109250854,0.000014630981,0.00006293458],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0002118783,0.00007110309,0.0001066048,0.00006000459,0.00017698303,0.000022207587,0.000055391007,0.00005836064,0.0018298547],"category_scores_gemma":[0.00020570512,0.000065884335,0.000018544795,0.00018253259,0.00009726879,0.00010048818,0.00003726515,0.00006708313,0.00010242475],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035920338,0.0006059041,0.021142485,0.00042760558,0.000018105602,0.000026428872,0.015353262,0.012231254,0.014919404,0.1578831,0.4596469,0.31738636],"study_design_scores_gemma":[0.0012629573,0.0006270821,0.023383351,0.0014462292,0.00007195567,0.000011344962,0.0040391847,0.14165783,0.014851837,0.5678428,0.24396932,0.0008361399],"about_ca_topic_score_codex":0.00081400527,"about_ca_topic_score_gemma":0.0038758218,"teacher_disagreement_score":0.6163791,"about_ca_system_score_codex":0.00014171888,"about_ca_system_score_gemma":0.000017715429,"threshold_uncertainty_score":0.9990826},"labels":[],"label_agreement":null},{"id":"W4414499338","doi":"10.1056/aidbp2500120","title":"Assessment of Large Language Models in Clinical Reasoning: A Novel Benchmarking Study","year":2025,"lang":"en","type":"article","venue":"NEJM AI","topic":"Clinical Reasoning and Diagnostic Skills","field":"Medicine","cited_by":28,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Benchmarking; Language model; Quality (philosophy); Identification (biology); Measure (data warehouse)","score_opus":0.037095345024797416,"score_gpt":0.45895407758977547,"score_spread":0.4218587325649781,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414499338","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97598404,0.00015430199,0.0041954103,0.0005577447,0.00028516655,0.00047571378,0.000009428488,0.000034082463,0.018304097],"genre_scores_gemma":[0.9949498,0.000025606976,0.002928823,0.001478453,0.000106156076,0.000025427427,0.000017320583,0.000010828687,0.00045754146],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9982597,0.00009676005,0.0007728121,0.00034451243,0.00028259013,0.00024361476],"domain_scores_gemma":[0.9948824,0.0043813894,0.000113913346,0.0004301701,0.00008817848,0.000103970546],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0018241976,0.000118778815,0.0005705086,0.00013437695,0.000025239138,0.000009989755,0.00010691305,0.0001307134,0.00006355077],"category_scores_gemma":[0.011510427,0.00010152879,0.00014454275,0.00035221962,0.000051654813,0.0000374901,0.00011672905,0.00047158616,0.000002757449],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008877677,0.0057655,0.9790066,0.000030334655,0.000088742054,0.00008244335,0.0005864494,0.00010608541,0.000057850906,0.0059099146,0.0008411314,0.0074362005],"study_design_scores_gemma":[0.0068835816,0.0006166971,0.9599111,0.0020498596,0.00013652132,0.0000025263796,0.0012838815,0.028689492,0.000017366641,0.00017744709,0.00014393147,0.0000875621],"about_ca_topic_score_codex":0.00028033066,"about_ca_topic_score_gemma":0.00013706814,"teacher_disagreement_score":0.028583407,"about_ca_system_score_codex":0.000053174314,"about_ca_system_score_gemma":0.00034271696,"threshold_uncertainty_score":0.99681604},"labels":[],"label_agreement":null},{"id":"W7117104805","doi":"10.1056/aioa2500522","title":"International Retrospective Observational Study of Continual Learning for AI on Endotracheal Tube Placement from Chest Radiographs","year":2025,"lang":"en","type":"article","venue":"NEJM AI","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Health Sciences Centre; Sunnybrook Health Science Centre; Western University; University of Toronto","funders":"","keywords":"Observational study; Radiography; Retrospective cohort study; Endotracheal tube; MEDLINE","score_opus":0.1655640285438163,"score_gpt":0.4549574241101151,"score_spread":0.2893933955662988,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7117104805","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9878873,0.00003303482,0.0015111038,0.007074291,0.0011007494,0.0008688031,0.000020449885,0.000028554261,0.0014756944],"genre_scores_gemma":[0.99684125,0.000012035979,0.00027741436,0.0014003597,0.0004423995,0.00014241837,0.000180644,0.000008638175,0.0006948202],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9989093,0.000040555326,0.00039666868,0.00024684044,0.00027192157,0.00013470529],"domain_scores_gemma":[0.9989159,0.00032470282,0.000101682446,0.00012699894,0.00048470235,0.00004599547],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00023869271,0.0000937747,0.0002074187,0.00016835392,0.00009853046,0.000014480776,0.00007653166,0.00006985161,0.00015951366],"category_scores_gemma":[0.0004481387,0.000091532595,0.00007085729,0.00018966723,0.000040893774,0.00006608142,0.000015152361,0.00026580485,0.0000055642486],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011824219,0.001201515,0.9655253,0.000030421777,0.00022677888,0.0000013634699,0.006482908,0.0002799592,0.0012525151,0.0015308466,0.004981388,0.01730456],"study_design_scores_gemma":[0.0010179623,0.0025658896,0.9481219,0.0001785904,0.00012152224,7.85487e-7,0.01661012,0.003831671,0.01466964,0.0018109648,0.010949624,0.00012133693],"about_ca_topic_score_codex":0.0010735417,"about_ca_topic_score_gemma":0.00013353248,"teacher_disagreement_score":0.01740343,"about_ca_system_score_codex":0.00018249129,"about_ca_system_score_gemma":0.00016564426,"threshold_uncertainty_score":0.3732591},"labels":[],"label_agreement":null}]}