{"meta":{"query_hash":"8f8a0ffd912f","filters":{"topic":"Multimodal Machine Learning Applications"},"cohort_total":476,"direct_labels_cover":3,"predictions_cover":476,"exported":476,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/8f8a0ffd912f","api":"https://metacan.xera.ac/api/v1/cohort?topic=Multimodal+Machine+Learning+Applications"},"results":[{"id":"W1479686495","doi":"10.1109/crv.2015.33","title":"Latent SVM for Object Localization in Weakly Labeled Videos","year":2015,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"Natural Sciences and Engineering Research Council of Canada; University of Manitoba; Nvidia","keywords":"Computer science; Object (grammar); Artificial intelligence; Support vector machine; The Internet; Computer vision; Class (philosophy); Pattern recognition (psychology); Information retrieval; World Wide Web","score_opus":0.03551878154791596,"score_gpt":0.3047090166105915,"score_spread":0.2691902350626756,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1479686495","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021669876,0.0004768055,0.9757413,0.0003685117,0.000052013525,0.00004092962,0.00031439358,0.00084657763,0.00048966566],"genre_scores_gemma":[0.7109454,0.0004995501,0.28080595,0.00039343315,0.00031745614,0.00021464123,0.00298,0.00017293815,0.0036704873],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987639,0.00048108696,0.00006171282,0.00034315925,0.0001984533,0.00015168593],"domain_scores_gemma":[0.99743736,0.001381458,0.00030648868,0.0002874503,0.00044226705,0.0001448602],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018381752,0.0010440209,0.001629035,0.00096005236,0.00050294737,0.0011302884,0.0016994664,0.0019356322,0.0019795916],"category_scores_gemma":[0.006863147,0.00035603583,0.0007568409,0.001437949,0.00074820226,0.0017673251,0.0013360948,0.0021960207,0.0011530821],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011044289,0.00068074476,0.0075406358,0.00053783646,0.00017293492,0.00037279935,0.00029385687,0.43354884,0.021733645,0.021844625,0.014060769,0.4981089],"study_design_scores_gemma":[0.000010052583,0.000029990157,0.00026468656,0.000008685743,0.000005849584,0.000017908622,0.000017397191,0.9926553,0.00085346866,0.005774605,0.0003570964,0.000005097497],"about_ca_topic_score_codex":0.0037460984,"about_ca_topic_score_gemma":0.0029480094,"teacher_disagreement_score":0.0037460984,"about_ca_system_score_codex":0.0008833538,"about_ca_system_score_gemma":0.0007790105,"threshold_uncertainty_score":0.009721339},"labels":[],"label_agreement":null},{"id":"W1492731187","doi":"","title":"Video Description Generation Incorporating Spatio-Temporal Features and a Soft-Attention Mechanism","year":2015,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":47,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Université de Sherbrooke; Université de Montréal","funders":"","keywords":"Computer science; Recurrent neural network; Mechanism (biology); Artificial intelligence; Frame (networking); Motion (physics); Long short term memory; Artificial neural network; Speech recognition; Machine learning; Natural language processing","score_opus":0.07850502296318435,"score_gpt":0.21182208747209735,"score_spread":0.133317064508913,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1492731187","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12941054,0.002009298,0.82295436,0.0012661231,0.00062470685,0.00058448507,0.0060451874,0.027133826,0.0099715125],"genre_scores_gemma":[0.57369256,0.00062928075,0.3877921,0.00049717334,0.0001953558,0.0003153281,0.02095099,0.0006333392,0.015293863],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99932396,0.00017335141,0.000053508054,0.00022101091,0.00016793207,0.00006017341],"domain_scores_gemma":[0.9982449,0.0007826963,0.00013174115,0.00038409972,0.0003597679,0.000096874144],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00094141066,0.001231959,0.00061038346,0.0011987066,0.00033633644,0.0010022632,0.0018762676,0.0011260691,0.0040852632],"category_scores_gemma":[0.0047858143,0.00038432726,0.00069817813,0.0010599338,0.0003352044,0.003178822,0.0012817206,0.0014888509,0.0018347864],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070733,0.0005239144,0.0024964472,0.00054809835,0.00012785484,0.00053139,0.00017898505,0.12306121,0.025742482,0.005220636,0.034069534,0.8067921],"study_design_scores_gemma":[0.000058572623,0.000107853455,0.0005222617,0.000018108596,0.000025677393,0.000115206116,0.000056569756,0.974506,0.01635754,0.0033025811,0.0049056453,0.000023934552],"about_ca_topic_score_codex":0.01101369,"about_ca_topic_score_gemma":0.016532773,"teacher_disagreement_score":0.01101369,"about_ca_system_score_codex":0.0012322976,"about_ca_system_score_gemma":0.0009923172,"threshold_uncertainty_score":0.021899164},"labels":[],"label_agreement":null},{"id":"W1508012366","doi":"10.1007/11414353_17","title":"Visual Capabilities in an Interactive Autonomous Robot","year":2006,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Robot; Human–computer interaction; Variety (cybernetics); Mobile robot; Embodied cognition; Gesture; Social robot; Context (archaeology); Artificial intelligence; Robot control","score_opus":0.011979985489256283,"score_gpt":0.28657387203121354,"score_spread":0.27459388654195727,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1508012366","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15615796,0.0024801218,0.44937056,0.0014633556,0.00018228809,0.000055652923,0.00014832204,0.0015353417,0.3886064],"genre_scores_gemma":[0.829723,0.0012538787,0.08753626,0.00013909707,0.000050733866,0.00011098673,0.000115088435,0.00015533301,0.08091555],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9999422,0.000009419614,0.000001847798,0.000013287979,0.000023382148,0.000009789349],"domain_scores_gemma":[0.9999256,0.000033662225,0.000006293253,0.0000068635345,0.000011076618,0.000016396163],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000066497385,0.00029115402,0.00014818198,0.00019418352,0.0004651754,0.0009847113,0.0006317614,0.0007301971,0.0061149746],"category_scores_gemma":[0.00027745156,0.0002698262,0.00025602742,0.00016300757,0.0010857288,0.001493729,0.0011661957,0.0005477411,0.0008011425],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019381163,0.00006218149,0.00035410066,0.00023967531,0.00002211163,0.0013135974,0.0021993855,0.04492591,0.10091118,0.7356921,0.0063876742,0.10769841],"study_design_scores_gemma":[0.000057856196,0.00029303753,0.0018393,0.00017523856,0.000043800108,0.0012606195,0.001564684,0.16906343,0.03371093,0.66310203,0.12878628,0.00010277642],"about_ca_topic_score_codex":0.0015641832,"about_ca_topic_score_gemma":0.001209081,"teacher_disagreement_score":0.0061149746,"about_ca_system_score_codex":0.00028870467,"about_ca_system_score_gemma":0.0002054703,"threshold_uncertainty_score":0.020456672},"labels":[],"label_agreement":null},{"id":"W1514535095","doi":"10.48550/arxiv.1502.03044","title":"Show, Attend and Tell: Neural Image Caption Generation with Visual Attention","year":2015,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":7525,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Institute for Advanced Research; University of Toronto; Université de Montréal","funders":"","keywords":"Computer science; Benchmark (surveying); Artificial intelligence; Visualization; Gaze; Object (grammar); Backpropagation; Salient; Sequence (biology); Object detection; Machine translation; Image (mathematics); Artificial neural network; Machine learning; Computer vision; Pattern recognition (psychology)","score_opus":0.06664802413473758,"score_gpt":0.20829710166008256,"score_spread":0.14164907752534497,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1514535095","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05071336,0.0031538953,0.8817399,0.0016083914,0.0010540169,0.00036739156,0.0022171808,0.04610902,0.013036763],"genre_scores_gemma":[0.52164805,0.0010229695,0.44794196,0.0014640845,0.00047966832,0.00044345402,0.005082408,0.0023500859,0.01956726],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996555,0.00009229153,0.000012298262,0.00012380324,0.00007257556,0.00004359007],"domain_scores_gemma":[0.99906975,0.00048255347,0.000048230413,0.00020882733,0.00013047906,0.000060163948],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008708236,0.0014295445,0.0006569051,0.00084006245,0.0004916052,0.0010312977,0.0024326087,0.0022285571,0.0075668497],"category_scores_gemma":[0.004161246,0.0005923039,0.0007824763,0.00078458106,0.0007864303,0.0025009317,0.0015885577,0.0019098796,0.0024481018],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006173333,0.0002694919,0.00097397546,0.0003818756,0.00018576268,0.00034405547,0.00029560237,0.15081361,0.0299698,0.016226262,0.0841597,0.71576256],"study_design_scores_gemma":[0.00005160049,0.00007078199,0.00022980326,0.000015455247,0.000022863429,0.00008817526,0.000024183717,0.96880853,0.011679035,0.014067011,0.0049213404,0.000021213971],"about_ca_topic_score_codex":0.007836649,"about_ca_topic_score_gemma":0.0113735525,"teacher_disagreement_score":0.007836649,"about_ca_system_score_codex":0.0010149954,"about_ca_system_score_gemma":0.00062398834,"threshold_uncertainty_score":0.025313616},"labels":[],"label_agreement":null},{"id":"W1566289585","doi":"10.1109/iccv.2015.11","title":"Aligning Books and Movies: Towards Story-Like Visual Explanations by Watching Movies and Reading Books","year":2015,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2068,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Institute for Advanced Research; University of Toronto","funders":"","keywords":"Computer science; Reading (process); Context (archaeology); Semantics (computer science); Sentence; Embedding; Object (grammar); Artificial intelligence; Feeling; Natural language processing; Character (mathematics); Linguistics; Psychology; History","score_opus":0.024420724676901304,"score_gpt":0.3090277660838301,"score_spread":0.2846070414069288,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1566289585","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19779623,0.0013266653,0.77929336,0.0013531669,0.00018407428,0.00020119149,0.0021212639,0.0059702154,0.011753821],"genre_scores_gemma":[0.6557407,0.0007571069,0.33062553,0.00032485652,0.00009229362,0.00009447573,0.003929224,0.00039647493,0.00803926],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99980277,0.000048352504,0.0000066898456,0.00009455799,0.00002992502,0.00001780736],"domain_scores_gemma":[0.9996352,0.00013805399,0.000057398793,0.00007241093,0.0000654471,0.00003138803],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00028341607,0.0007629422,0.0002569709,0.0005202683,0.00027834368,0.00090757903,0.00074578053,0.00084174547,0.004448122],"category_scores_gemma":[0.0022904682,0.0003593072,0.00063833245,0.00053751585,0.00041591187,0.0023937959,0.0008487474,0.001176413,0.0009780509],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006945355,0.00030329943,0.015359844,0.0005684051,0.00027100684,0.0009039085,0.003014842,0.07789803,0.0741183,0.034980662,0.026535407,0.7653518],"study_design_scores_gemma":[0.000038743477,0.00015552125,0.0125357015,0.000097678014,0.0001228415,0.0005327721,0.0009875154,0.89412165,0.029746313,0.037976652,0.023631042,0.000053568732],"about_ca_topic_score_codex":0.0047274916,"about_ca_topic_score_gemma":0.008455875,"teacher_disagreement_score":0.0047274916,"about_ca_system_score_codex":0.00047742383,"about_ca_system_score_gemma":0.00039186375,"threshold_uncertainty_score":0.014880419},"labels":[],"label_agreement":null},{"id":"W1575833922","doi":"10.48550/arxiv.1505.02074","title":"Exploring Models and Data for Image Question Answering","year":2015,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":390,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Image (mathematics); Question answering; Suite; Artificial intelligence; Segmentation; Image segmentation; Simple (philosophy); Object (grammar); Baseline (sea); Information retrieval; Pattern recognition (psychology); Machine learning; Data mining","score_opus":0.4143372677061843,"score_gpt":0.2670283867265147,"score_spread":0.1473088809796696,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1575833922","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09592718,0.008450113,0.85918576,0.009577334,0.00038974025,0.00054116396,0.0147803845,0.006618079,0.0045301947],"genre_scores_gemma":[0.46218687,0.0017458671,0.47880423,0.0019636482,0.0004675657,0.0009868657,0.050531194,0.00048057706,0.0028332113],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9925256,0.0040088017,0.0003584115,0.0021079641,0.0007609351,0.00023838755],"domain_scores_gemma":[0.9751848,0.017537277,0.00096601533,0.004257829,0.0015767872,0.00047725465],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008460967,0.0022985723,0.0016972471,0.0033730087,0.0009894086,0.0036240795,0.0041685337,0.0041264286,0.004163468],"category_scores_gemma":[0.039146904,0.0008524997,0.0028330241,0.0028465712,0.0019402021,0.011135656,0.0046235453,0.005614033,0.0028913252],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016333609,0.0018476015,0.023815034,0.0025942456,0.00061158865,0.0003894912,0.0014201525,0.28783417,0.011795067,0.057648655,0.07647693,0.5339337],"study_design_scores_gemma":[0.000064535976,0.00015188805,0.0012984594,0.000091470436,0.000046035697,0.00012331287,0.00025529636,0.8974559,0.0027816256,0.08701441,0.01067635,0.000040686922],"about_ca_topic_score_codex":0.008511794,"about_ca_topic_score_gemma":0.00939417,"teacher_disagreement_score":0.008511794,"about_ca_system_score_codex":0.0031810831,"about_ca_system_score_gemma":0.001548955,"threshold_uncertainty_score":0.0447464},"labels":[],"label_agreement":null},{"id":"W1581862991","doi":"10.1109/tpami.2015.2505297","title":"Saying What You're Looking For: Linguistics Meets Video Search","year":2015,"lang":"en","type":"article","venue":"IEEE Transactions on Pattern Analysis and Machine Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Army Research Laboratory; Computer Science and Artificial Intelligence Laboratory, Massachusetts Institute of Technology; University of Waterloo; Purdue University","keywords":"Computer science; Natural language processing; Parsing; Sentence; Artificial intelligence; Meaning (existential); Semantics (computer science); Noun; Natural language; Object (grammar); Information retrieval; Linguistics; Psychology","score_opus":0.049082483547532654,"score_gpt":0.33033742457999854,"score_spread":0.2812549410324659,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1581862991","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22782773,0.022718953,0.59360486,0.028134739,0.0004771748,0.00092325115,0.008901155,0.005390559,0.112021565],"genre_scores_gemma":[0.66100395,0.004947778,0.31369683,0.002097936,0.0006877244,0.0004559652,0.0058036963,0.00059985527,0.010706297],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99767417,0.001151959,0.00015817102,0.0005009423,0.00038803957,0.00012673439],"domain_scores_gemma":[0.9931431,0.004872805,0.0005967018,0.00044941655,0.00067238096,0.0002656212],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019516263,0.0008092009,0.0007950084,0.007303797,0.0018288594,0.0041534994,0.0010168885,0.0018409586,0.008421016],"category_scores_gemma":[0.0135298455,0.00051270536,0.00062374247,0.0062224097,0.0021240443,0.012433489,0.002266425,0.0011789955,0.0030472728],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007223112,0.0003065276,0.03305429,0.0025359269,0.00015030349,0.002048571,0.018663244,0.004034195,0.03992159,0.119793326,0.05234706,0.7264226],"study_design_scores_gemma":[0.00017744745,0.0005067045,0.0687783,0.00083865144,0.00028841075,0.006501259,0.04277359,0.15850711,0.0239487,0.40186462,0.29534376,0.00047153232],"about_ca_topic_score_codex":0.013011987,"about_ca_topic_score_gemma":0.020813618,"teacher_disagreement_score":0.013011987,"about_ca_system_score_codex":0.002190957,"about_ca_system_score_gemma":0.0014516478,"threshold_uncertainty_score":0.028171062},"labels":[],"label_agreement":null},{"id":"W1586939924","doi":"10.1109/iccv.2015.512","title":"Describing Videos by Exploiting Temporal Structure","year":2015,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":956,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Université de Montréal; Université de Sherbrooke","funders":"Compute Canada; Canada Research Chairs; Canadian Institute for Advanced Research","keywords":"Computer science; Recurrent neural network; Artificial intelligence; Representation (politics); Convolutional neural network; Context (archaeology); Motion (physics); Natural language; Pattern recognition (psychology); Artificial neural network; Machine learning","score_opus":0.05364504113686984,"score_gpt":0.2721015833326518,"score_spread":0.21845654219578192,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1586939924","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13811171,0.004737106,0.8219122,0.0010809525,0.00025508108,0.00044762995,0.013419471,0.0061815693,0.013854241],"genre_scores_gemma":[0.6084393,0.0036439446,0.34403953,0.00039014843,0.00021988229,0.00031799742,0.03174729,0.0004741719,0.010727631],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99936897,0.00016211119,0.000060472856,0.00018426031,0.0001765188,0.000047568832],"domain_scores_gemma":[0.9984493,0.00060937804,0.00029327333,0.00030980833,0.00028364692,0.00005449543],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00070773775,0.0011014097,0.00047003763,0.00220224,0.00035361032,0.0010248086,0.0010459811,0.0006807908,0.0022486055],"category_scores_gemma":[0.004547385,0.00028947476,0.0007594549,0.001981998,0.00037840498,0.0037867455,0.0009460966,0.0008668533,0.0009434641],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00080336997,0.00020754563,0.0081413975,0.001016046,0.00022107553,0.0007949208,0.0006058461,0.124450706,0.027029755,0.026535975,0.03079319,0.77940005],"study_design_scores_gemma":[0.000038715938,0.00015709815,0.0030889844,0.000117861884,0.00012355801,0.0005851814,0.0002636113,0.9158756,0.021468403,0.024398379,0.033817887,0.00006479347],"about_ca_topic_score_codex":0.018660065,"about_ca_topic_score_gemma":0.025964314,"teacher_disagreement_score":0.018660065,"about_ca_system_score_codex":0.0013257433,"about_ca_system_score_gemma":0.0007407964,"threshold_uncertainty_score":0.037102938},"labels":[],"label_agreement":null},{"id":"W1794939664","doi":"10.48550/arxiv.1312.6171","title":"Learning Paired-associate Images with An Unsupervised Deep Learning Architecture","year":2013,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Acadia University","funders":"","keywords":"MNIST database; Computer science; Associative property; Artificial intelligence; Restricted Boltzmann machine; Unsupervised learning; Content-addressable memory; Representation (politics); Boltzmann machine; Deep learning; Pattern recognition (psychology); Modal; Artificial neural network; Machine learning; Mathematics","score_opus":0.028708464196324054,"score_gpt":0.185583910849351,"score_spread":0.15687544665302694,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1794939664","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06634902,0.00013250671,0.92916536,0.00018411143,0.000041192467,0.000034511402,0.000081817,0.0015996833,0.0024117392],"genre_scores_gemma":[0.72188944,0.00009533901,0.2725181,0.00019601139,0.000029033861,0.000089795,0.0001955103,0.00008816274,0.0048986454],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99973387,0.00007019932,0.000009506806,0.000101695994,0.00005561464,0.000029069917],"domain_scores_gemma":[0.99957997,0.00014704239,0.000051321735,0.00010711558,0.00008453176,0.000029984303],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000635557,0.0005458934,0.0004155302,0.0002992438,0.00022741276,0.00059032236,0.001208868,0.0008091319,0.0019097959],"category_scores_gemma":[0.0016316887,0.00037967708,0.0005924846,0.0003446031,0.00067638315,0.0014398716,0.0011549955,0.0011739028,0.0005677785],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003101365,0.00032717906,0.0032170706,0.00013071613,0.0002029845,0.0001906712,0.00027202172,0.5024809,0.055226427,0.02604159,0.0034826135,0.40811777],"study_design_scores_gemma":[0.000004668261,0.00003585722,0.00019275928,0.0000038788153,0.000009086701,0.000027119679,0.000009763885,0.9847211,0.0064741666,0.008106253,0.00040955527,0.0000057183533],"about_ca_topic_score_codex":0.0012928144,"about_ca_topic_score_gemma":0.0024045876,"teacher_disagreement_score":0.0019097959,"about_ca_system_score_codex":0.0004697767,"about_ca_system_score_gemma":0.0004144463,"threshold_uncertainty_score":0.0063889027},"labels":[],"label_agreement":null},{"id":"W1956526898","doi":"10.48550/arxiv.1509.06812","title":"Learning Wake-Sleep Recurrent Attention Models","year":2015,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Inference; Convolutional neural network; Artificial intelligence; Machine learning; Stochastic gradient descent; Deep learning; Variance (accounting); Stochastic process; Artificial neural network; Mathematics; Statistics","score_opus":0.09991577746164898,"score_gpt":0.20594045657521004,"score_spread":0.10602467911356106,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1956526898","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0742817,0.00060316635,0.9180798,0.0007321992,0.000104131504,0.000044573353,0.00039090152,0.002061178,0.0037022557],"genre_scores_gemma":[0.89301217,0.00036479512,0.0958055,0.0004360005,0.00012548722,0.0001352368,0.0009206983,0.0002524516,0.008947626],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997341,0.00006740575,0.000009492369,0.00009781988,0.000038493657,0.000052583793],"domain_scores_gemma":[0.9993487,0.00035019446,0.000072361814,0.00007605151,0.00011301252,0.000039687166],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000790248,0.0007741864,0.00068377826,0.00048555533,0.00030962774,0.00072022143,0.001542941,0.0010552503,0.0026361772],"category_scores_gemma":[0.0033748383,0.00070871064,0.0008119506,0.0004847853,0.000586818,0.0014796116,0.0011530536,0.0014829166,0.0006170533],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001950902,0.00012696093,0.0030323826,0.000085700674,0.000114033675,0.000173636,0.00019842187,0.7672042,0.007769427,0.032362077,0.00907461,0.17966338],"study_design_scores_gemma":[0.000003810288,0.0000076325805,0.00014549932,0.000002501249,0.000005374249,0.000007740963,0.000003408469,0.9927475,0.00045400337,0.006391822,0.00022770179,0.0000030901267],"about_ca_topic_score_codex":0.007859056,"about_ca_topic_score_gemma":0.015796803,"teacher_disagreement_score":0.007859056,"about_ca_system_score_codex":0.00082790316,"about_ca_system_score_gemma":0.00068478973,"threshold_uncertainty_score":0.01562661},"labels":[],"label_agreement":null},{"id":"W1982527966","doi":"10.1145/1054972.1055042","title":"A visual recipe book for persons with language impairments","year":2005,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Recipe; Computer science; Presentation (obstetrics); Aphasia; Visual language; Natural language processing; Modal; Artificial intelligence; Human–computer interaction; Linguistics; Psychology; Cognitive psychology; Medicine","score_opus":0.006463165865420177,"score_gpt":0.29848421595249364,"score_spread":0.29202105008707346,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1982527966","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3485536,0.004374714,0.41843066,0.0059240283,0.0008339508,0.0023963228,0.0086155385,0.08210137,0.12876995],"genre_scores_gemma":[0.3542775,0.0027579984,0.5150011,0.0018478503,0.00013840174,0.0010346845,0.0068629947,0.0015869391,0.116492495],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9999379,0.000021410317,0.0000061657242,0.000010938407,0.000016545793,0.0000070344367],"domain_scores_gemma":[0.99956566,0.0002357306,0.000021100777,0.000053923744,0.00006142425,0.0000622156],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00025575148,0.0006272126,0.00024968703,0.00032219497,0.00024428335,0.0004555178,0.000599745,0.0006950943,0.038095627],"category_scores_gemma":[0.002082375,0.000119205884,0.00034852768,0.00016280702,0.00016368368,0.0009718386,0.00079535384,0.00040229247,0.011720026],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007048353,0.0007970828,0.0037713503,0.0011904882,0.00004509958,0.0026577534,0.0010981034,0.0010254979,0.047523987,0.0017287581,0.13689098,0.8025661],"study_design_scores_gemma":[0.0007019603,0.0034900282,0.038862392,0.0009085652,0.00024576465,0.030965725,0.0031126654,0.03156058,0.094485216,0.014289389,0.7810122,0.00036556475],"about_ca_topic_score_codex":0.00049956766,"about_ca_topic_score_gemma":0.0019472907,"teacher_disagreement_score":0.038095627,"about_ca_system_score_codex":0.00011250433,"about_ca_system_score_gemma":0.0002501114,"threshold_uncertainty_score":0.12744254},"labels":[],"label_agreement":null},{"id":"W1995820507","doi":"10.1109/cvpr.2013.340","title":"A Thousand Frames in Just a Few Words: Lingual Description of Videos through Latent Topics and Sparse Object Stitching","year":2013,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":294,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Interior Business Center; Defense Advanced Research Projects Agency; U.S. Department of the Interior; Simon Fraser University; National Science Foundation","keywords":"Computer science; Image stitching; Artificial intelligence; Natural language processing; Natural language; Object (grammar); Probabilistic logic; Image (mathematics); Language model; Annotation; Natural language generation; Information retrieval; Pattern recognition (psychology)","score_opus":0.03467305324326582,"score_gpt":0.29148073092247134,"score_spread":0.25680767767920554,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1995820507","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08455831,0.0008990722,0.9041079,0.0006911561,0.00010034549,0.00016209272,0.0014977608,0.0047812844,0.0032020193],"genre_scores_gemma":[0.4743637,0.00086057215,0.51457024,0.00032777325,0.00015228553,0.00020806906,0.00429937,0.0004751237,0.0047428887],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994306,0.00018780182,0.000036672576,0.00015708979,0.00013395556,0.00005389682],"domain_scores_gemma":[0.99867564,0.00048061708,0.00016675677,0.00032330063,0.0002893141,0.00006428408],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009175241,0.0006201326,0.00049563375,0.0016296919,0.00044243605,0.001042037,0.0009010676,0.00069918425,0.0016336062],"category_scores_gemma":[0.005076235,0.0003328977,0.00072797894,0.0012370003,0.0005681436,0.0025176255,0.0011750706,0.0010997562,0.00087089086],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008319698,0.000170485,0.006583968,0.00035328852,0.00013654378,0.000805636,0.001820062,0.04197271,0.047419667,0.018282523,0.01855282,0.86307025],"study_design_scores_gemma":[0.000030767376,0.00013266958,0.0043310863,0.000079652586,0.00008725578,0.0006566847,0.00071143056,0.9229225,0.029945917,0.0246462,0.016371395,0.00008444411],"about_ca_topic_score_codex":0.008098359,"about_ca_topic_score_gemma":0.009755483,"teacher_disagreement_score":0.008098359,"about_ca_system_score_codex":0.00073952787,"about_ca_system_score_gemma":0.00063945074,"threshold_uncertainty_score":0.016102433},"labels":[],"label_agreement":null},{"id":"W1999830516","doi":"10.4000/lettre-cdf.86","title":"Modèles computationnels du mouvement humain","year":2009,"lang":"fr","type":"article","venue":"La lettre du Collège de France","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Humanities; Art","score_opus":0.01060716546016197,"score_gpt":0.2507894383769145,"score_spread":0.2401822729167525,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1999830516","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05889198,0.002897959,0.9090545,0.0019959756,0.00033099804,0.00007461371,0.00040529668,0.00083260436,0.025516022],"genre_scores_gemma":[0.853026,0.003373688,0.09817406,0.00028190727,0.00014379465,0.00025154112,0.0004932224,0.00023768285,0.04401816],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997929,0.00004346782,0.00000689035,0.000056154036,0.000068846195,0.000031771087],"domain_scores_gemma":[0.99952924,0.00023937722,0.000036908197,0.00003831307,0.0001135901,0.000042626765],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004238922,0.0005892141,0.0007192588,0.00046117947,0.00060959655,0.0018706413,0.0009681613,0.0010483654,0.0043625003],"category_scores_gemma":[0.0020474,0.00047924262,0.0009920596,0.0005192435,0.0011730901,0.0018260899,0.001127798,0.0010117139,0.000779258],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019397684,0.000032872806,0.0016515164,0.00017800214,0.0000742555,0.00021680107,0.00038361622,0.7678482,0.009325369,0.18340643,0.0040069665,0.032682065],"study_design_scores_gemma":[0.000012822166,0.000026290993,0.0005001673,0.000023309303,0.0000138162995,0.000050858434,0.00005295415,0.95897657,0.0013817039,0.032497298,0.006438367,0.000025843528],"about_ca_topic_score_codex":0.028610291,"about_ca_topic_score_gemma":0.017397285,"teacher_disagreement_score":0.028610291,"about_ca_system_score_codex":0.0014206871,"about_ca_system_score_gemma":0.0014784749,"threshold_uncertainty_score":0.056887507},"labels":[],"label_agreement":null},{"id":"W2048343491","doi":"10.1109/cvpr.2014.455","title":"What Are You Talking About? Text-to-Image Coreference","year":2014,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":205,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Coreference; Computer science; Artificial intelligence; Parsing; Natural language processing; Exploit; Object (grammar); Image (mathematics); Noun; Resolution (logic)","score_opus":0.014036303612288489,"score_gpt":0.27823979989435704,"score_spread":0.2642034962820686,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2048343491","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33776864,0.009787598,0.54745036,0.011348024,0.0020074768,0.00041542956,0.022253033,0.005746095,0.063223325],"genre_scores_gemma":[0.8142791,0.0017870659,0.1511826,0.004530261,0.00040204654,0.0003015937,0.011728458,0.00085304695,0.014935846],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981688,0.00066323305,0.000118168624,0.00064185413,0.0002840967,0.00012374994],"domain_scores_gemma":[0.9963431,0.0022833687,0.00040746675,0.00038808954,0.0005016981,0.00007618854],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015065373,0.0006858076,0.000611625,0.0016753793,0.0012226886,0.0015271156,0.0008530335,0.0015954054,0.0051254258],"category_scores_gemma":[0.009573392,0.00030358206,0.00059597334,0.0019350705,0.00084182975,0.003588327,0.0020111755,0.0014479933,0.0027242883],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016816535,0.00017409862,0.035984144,0.0021308877,0.00026422954,0.0034744793,0.018811734,0.0057349415,0.04548193,0.051896717,0.17354538,0.6608197],"study_design_scores_gemma":[0.00007735655,0.0002504037,0.036416005,0.00079056644,0.00041398927,0.007783058,0.02242094,0.08773324,0.10346587,0.14251466,0.5978267,0.00030711686],"about_ca_topic_score_codex":0.0034077503,"about_ca_topic_score_gemma":0.0047833305,"teacher_disagreement_score":0.0051254258,"about_ca_system_score_codex":0.0009824544,"about_ca_system_score_gemma":0.0005207406,"threshold_uncertainty_score":0.01714629},"labels":[],"label_agreement":null},{"id":"W2110380922","doi":"10.1109/icra.2014.6906913","title":"Curiosity based exploration for learning terrain models","year":2014,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Terrain; Perplexity; Discriminative model; Path (computing); Visualization; Plan (archaeology)","score_opus":0.030668714280670105,"score_gpt":0.28077977677587596,"score_spread":0.25011106249520587,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2110380922","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14192158,0.0011833616,0.8510492,0.0006634734,0.000037389582,0.00006785368,0.0008021524,0.0016102436,0.0026647442],"genre_scores_gemma":[0.9112957,0.0005229366,0.083731405,0.00017520442,0.00006149415,0.00014739064,0.0018876999,0.0001765688,0.002001637],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995784,0.00013348756,0.00001893162,0.0001659441,0.000059121587,0.000044079094],"domain_scores_gemma":[0.998264,0.0012025139,0.00016792762,0.00019407534,0.00009594855,0.0000755967],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00090447307,0.0010103836,0.0009211031,0.0012070657,0.00048263554,0.00079077046,0.0013279079,0.00076110853,0.0019940957],"category_scores_gemma":[0.00513256,0.0005913337,0.0011638857,0.0009971529,0.000831104,0.0018505218,0.0013610674,0.0013397206,0.0005289135],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037758224,0.000102155,0.009613232,0.00029953694,0.00029962923,0.00018825446,0.00068247505,0.7745041,0.0057918755,0.019780817,0.0041434593,0.18421681],"study_design_scores_gemma":[0.000016557304,0.000041854026,0.00069888093,0.000013558861,0.00001609169,0.00005815779,0.000027529073,0.97455084,0.00073632534,0.02311054,0.000718593,0.000011046165],"about_ca_topic_score_codex":0.0034495576,"about_ca_topic_score_gemma":0.005131065,"teacher_disagreement_score":0.0034495576,"about_ca_system_score_codex":0.0007943091,"about_ca_system_score_gemma":0.0005357734,"threshold_uncertainty_score":0.006858945},"labels":[],"label_agreement":null},{"id":"W2111141593","doi":"10.1109/iccv.2007.4408877","title":"Learning Structured Appearance Models from Captioned Images of Cluttered Scenes","year":2007,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Artificial intelligence; Feature (linguistics); ENCODE; Computer vision; Object (grammar); Graph; Pattern recognition (psychology)","score_opus":0.012134812412102477,"score_gpt":0.26091229373609864,"score_spread":0.24877748132399616,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2111141593","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12855212,0.0004951116,0.8616598,0.00031463904,0.0001266676,0.00024861336,0.0011761637,0.005499383,0.0019275703],"genre_scores_gemma":[0.62776005,0.0004592928,0.3592835,0.0004026198,0.0001655483,0.0003243777,0.00772335,0.0005300039,0.00335123],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99920195,0.00021467001,0.000026511028,0.00037867978,0.000105534724,0.000072630755],"domain_scores_gemma":[0.9972677,0.0013211252,0.00026394697,0.00056728354,0.00042390235,0.00015608185],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00094854855,0.0018905965,0.0013326355,0.0015819534,0.00045395314,0.0014718398,0.0027055226,0.0024222108,0.001809559],"category_scores_gemma":[0.005081295,0.0010729751,0.0015423716,0.0015746669,0.0009870944,0.0021427302,0.00089352357,0.0025473177,0.0012997091],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010076463,0.00062154105,0.0060696215,0.00048287906,0.00036277133,0.0011765185,0.0005765747,0.5668203,0.041144475,0.004919944,0.019184416,0.3576334],"study_design_scores_gemma":[0.00001547416,0.00004927183,0.00047536002,0.000010124594,0.000016701217,0.00009379033,0.000035054203,0.99468094,0.0020832638,0.0019987044,0.00053056335,0.000010617761],"about_ca_topic_score_codex":0.004229298,"about_ca_topic_score_gemma":0.00762704,"teacher_disagreement_score":0.004229298,"about_ca_system_score_codex":0.0009585731,"about_ca_system_score_gemma":0.00047272217,"threshold_uncertainty_score":0.008409381},"labels":[],"label_agreement":null},{"id":"W2111407473","doi":"10.1109/crv.2012.37","title":"Learning Categorical Shape from Captioned Images","year":2012,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Artificial intelligence; Computer science; Minimum bounding box; Object (grammar); Bounding overwatch; Computer vision; Boundary (topology); Active appearance model; Annotation; Set (abstract data type); Categorical variable; Pattern recognition (psychology); Image (mathematics); Class (philosophy); Cognitive neuroscience of visual object recognition; Mathematics; Machine learning","score_opus":0.014716694765619813,"score_gpt":0.26916058934433645,"score_spread":0.25444389457871663,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2111407473","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16926056,0.000571726,0.81389546,0.00032858842,0.000166861,0.0002482308,0.0024163532,0.008968299,0.0041438336],"genre_scores_gemma":[0.65343845,0.00043211534,0.3298505,0.0003918731,0.00010902771,0.00021378575,0.011970596,0.0004193981,0.0031742814],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992803,0.000120582634,0.00002562492,0.0003590028,0.00013050433,0.000083964565],"domain_scores_gemma":[0.997785,0.000700591,0.00023137612,0.0006754843,0.00044842652,0.00015912473],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006956215,0.0014964716,0.0011738826,0.0017755497,0.00037723637,0.001616022,0.0022880156,0.0020941244,0.0029211447],"category_scores_gemma":[0.0042348397,0.00072167284,0.0014863103,0.0018923375,0.000879075,0.002307112,0.0012678427,0.0016237305,0.002033214],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00096122775,0.00028237738,0.0071509434,0.00042009956,0.00019620203,0.0008972496,0.00042540635,0.16687535,0.06584552,0.005988043,0.013911911,0.73704576],"study_design_scores_gemma":[0.000023251952,0.00015657865,0.0020391406,0.000029930101,0.00002335291,0.00025249593,0.00013800449,0.97676176,0.010294859,0.007914358,0.0023328927,0.00003341244],"about_ca_topic_score_codex":0.0036395192,"about_ca_topic_score_gemma":0.0062712813,"teacher_disagreement_score":0.0036395192,"about_ca_system_score_codex":0.0010741303,"about_ca_system_score_gemma":0.0005314859,"threshold_uncertainty_score":0.0097721815},"labels":[],"label_agreement":null},{"id":"W2113005879","doi":"10.1109/iccv.2013.462","title":"Handling Uncertain Tags in Visual Recognition","year":2013,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Training set; Artificial intelligence; Noise (video); Set (abstract data type); Noisy data; Pattern recognition (psychology); Data set; Computer vision; Machine learning; Image (mathematics)","score_opus":0.019587749397098838,"score_gpt":0.2892064570180142,"score_spread":0.26961870762091533,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2113005879","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.048617963,0.0020637305,0.9446949,0.0007070023,0.00015894549,0.00006579477,0.00037456123,0.0014980292,0.0018189993],"genre_scores_gemma":[0.7120379,0.0012499972,0.27902377,0.0009438651,0.00037318643,0.00018296717,0.0018671552,0.00040417665,0.003916972],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.995353,0.0012496993,0.0002814568,0.0014418885,0.0013293322,0.0003446505],"domain_scores_gemma":[0.9872938,0.0075653214,0.00092829415,0.0025481426,0.0014205298,0.00024388629],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046096114,0.0012235414,0.001597828,0.0019467014,0.0013595395,0.0028389413,0.0024702766,0.0030017206,0.0014381277],"category_scores_gemma":[0.02905089,0.00079739606,0.0007985142,0.0022560325,0.0027254026,0.0056342394,0.003398185,0.0028416812,0.0012040494],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007277897,0.00014042426,0.009136014,0.00047562172,0.00013227064,0.00076104957,0.0008420235,0.21826209,0.01707727,0.028440975,0.012905961,0.71109843],"study_design_scores_gemma":[0.00002584159,0.00009461462,0.0022004622,0.00009822501,0.000044006847,0.00058484066,0.000301616,0.8650441,0.02282732,0.10201178,0.00669624,0.00007102952],"about_ca_topic_score_codex":0.0042224894,"about_ca_topic_score_gemma":0.004345828,"teacher_disagreement_score":0.0046096114,"about_ca_system_score_codex":0.0015508307,"about_ca_system_score_gemma":0.0009031357,"threshold_uncertainty_score":0.02437824},"labels":[],"label_agreement":null},{"id":"W2130844143","doi":"10.13140/2.1.1135.8081","title":"Unsupervised Disambiguation of Image Captions","year":2012,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Latent Dirichlet allocation; Artificial intelligence; Set (abstract data type); sort; Natural language processing; Construct (python library); Context (archaeology); Word (group theory); Word-sense disambiguation; Image (mathematics); Topic model; Pattern recognition (psychology); Information retrieval","score_opus":0.01478172752583834,"score_gpt":0.28339435324562423,"score_spread":0.2686126257197859,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2130844143","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.30025217,0.0043984563,0.64689124,0.0012889483,0.0012481746,0.0008741913,0.007816243,0.009766396,0.027464164],"genre_scores_gemma":[0.5753178,0.00077086996,0.39992708,0.00050428294,0.0005590406,0.0005526833,0.015459364,0.0010723766,0.0058365497],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974125,0.00070965255,0.00015349216,0.0012119703,0.00034767424,0.00016469126],"domain_scores_gemma":[0.9950964,0.0020377499,0.0005142921,0.00096951553,0.0011940625,0.00018805431],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015879982,0.0015347945,0.0012606033,0.0051576607,0.0017371608,0.0021552008,0.0019407942,0.0017458426,0.0040848134],"category_scores_gemma":[0.010347041,0.0006782218,0.0011069867,0.0037518633,0.0017235137,0.0037794185,0.0023729617,0.001738736,0.002595312],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019566733,0.00063496054,0.012280123,0.0017461515,0.00040660068,0.0021175283,0.0030059263,0.04345054,0.1271718,0.035561487,0.082809,0.68885916],"study_design_scores_gemma":[0.00023062176,0.00034744118,0.028565671,0.00034469416,0.00030462348,0.002852107,0.0027995573,0.61277795,0.16735338,0.08130664,0.102800265,0.00031710087],"about_ca_topic_score_codex":0.0033590614,"about_ca_topic_score_gemma":0.0066068843,"teacher_disagreement_score":0.0051576607,"about_ca_system_score_codex":0.0010275498,"about_ca_system_score_gemma":0.0009969396,"threshold_uncertainty_score":0.01366508},"labels":[],"label_agreement":null},{"id":"W2155292833","doi":"10.48550/arxiv.1511.02793","title":"Generating Images from Captions with Attention","year":2015,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":75,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Generative grammar; Computer science; Artificial intelligence; Generative model; Baseline (sea); Image (mathematics); Natural language processing; Training set; Natural (archaeology); Natural language generation; Machine learning; Natural language; Geography","score_opus":0.06370869466700849,"score_gpt":0.20355322829454486,"score_spread":0.13984453362753638,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2155292833","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036546756,0.0013538003,0.9371267,0.0010066031,0.00032415224,0.00042990074,0.0013717089,0.011543063,0.010297334],"genre_scores_gemma":[0.5303128,0.0010929238,0.4418806,0.0012388749,0.00031187947,0.0006369305,0.006775067,0.0023063372,0.015444648],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99938786,0.00018019954,0.000020806225,0.00023446146,0.000113414375,0.000063216125],"domain_scores_gemma":[0.99853075,0.0008035674,0.000079366684,0.0003247025,0.00018850548,0.000073001334],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00085519394,0.0016202026,0.0008777476,0.0011029731,0.00041997485,0.0015203586,0.002007437,0.0018550218,0.0073502855],"category_scores_gemma":[0.0048770695,0.00085040316,0.0015465793,0.0008809328,0.0009918096,0.0017148417,0.0012408403,0.0019742318,0.002568982],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005438644,0.00030593615,0.0019487402,0.0006642641,0.00018370239,0.000775437,0.0005881962,0.4984543,0.03407443,0.036210928,0.054962438,0.37128776],"study_design_scores_gemma":[0.000036335347,0.0000458284,0.00020893304,0.00002332752,0.000020263167,0.00017595402,0.000030790627,0.9753677,0.007032532,0.012175461,0.0048647695,0.000018069244],"about_ca_topic_score_codex":0.005442805,"about_ca_topic_score_gemma":0.007964741,"teacher_disagreement_score":0.0073502855,"about_ca_system_score_codex":0.0013224857,"about_ca_system_score_gemma":0.00068099296,"threshold_uncertainty_score":0.024589181},"labels":[],"label_agreement":null},{"id":"W2160257818","doi":"10.3115/1119212.1119220","title":"Why can't José read?","year":2003,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Artificial intelligence; Object (grammar); Set (abstract data type); Process (computing); Context (archaeology); Mobile robot; Robot; Image (mathematics); Computer vision; Cognitive neuroscience of visual object recognition; Spatial contextual awareness; Pattern recognition (psychology)","score_opus":0.012240741510052246,"score_gpt":0.2586873306050244,"score_spread":0.24644658909497216,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2160257818","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20891006,0.021973558,0.31611466,0.20197782,0.006135439,0.00023803329,0.0028010719,0.0022645707,0.23958476],"genre_scores_gemma":[0.7584183,0.008948209,0.09242607,0.013211995,0.0019750362,0.00012506433,0.0018308287,0.0006487376,0.122415714],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992424,0.00021109964,0.00002586818,0.00028117976,0.00012803047,0.00011147374],"domain_scores_gemma":[0.995365,0.0030840242,0.000441082,0.0002979613,0.0004394156,0.00037252106],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011979709,0.00042153936,0.0005298965,0.00046577462,0.0022927315,0.0034120898,0.0012481647,0.0028454305,0.015683277],"category_scores_gemma":[0.010623182,0.00030719666,0.00042303198,0.0011030786,0.0028897584,0.0073842215,0.0011335257,0.0019321083,0.00585815],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00064530934,0.00029400788,0.0136241205,0.00088337983,0.00010233477,0.0026337951,0.0059249145,0.010232862,0.003522502,0.34322432,0.22712909,0.39178336],"study_design_scores_gemma":[0.00005746243,0.0001956134,0.0042480333,0.00017786359,0.00007927508,0.0019319593,0.006459979,0.042252637,0.0036952705,0.5831026,0.3576942,0.000105237385],"about_ca_topic_score_codex":0.008447344,"about_ca_topic_score_gemma":0.013067377,"teacher_disagreement_score":0.015683277,"about_ca_system_score_codex":0.00084054447,"about_ca_system_score_gemma":0.00081265037,"threshold_uncertainty_score":0.052465737},"labels":[],"label_agreement":null},{"id":"W2171361956","doi":"","title":"Multimodal Neural Language Models","year":2014,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":568,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Canadian Institute for Advanced Research","funders":"","keywords":"Computer science; Artificial intelligence; Convolutional neural network; Natural language processing; Modalities; Focus (optics); Sentence; Phrase; Natural language; Image (mathematics); Artificial neural network; Language model; Word (group theory); Speech recognition; Linguistics","score_opus":0.010895750074746786,"score_gpt":0.26624760043780255,"score_spread":0.25535185036305574,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2171361956","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0653773,0.002057777,0.8965693,0.0038786584,0.00044242814,0.00015676308,0.0030372615,0.0034656888,0.025014786],"genre_scores_gemma":[0.86595434,0.0012005105,0.09123997,0.0009465327,0.00035399344,0.00035274585,0.0026265883,0.0003660118,0.036959358],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994388,0.00018245961,0.000030719577,0.00018052281,0.000099324716,0.00006811745],"domain_scores_gemma":[0.9988134,0.00066970324,0.00010854512,0.0001455728,0.00020995196,0.000052815776],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00092776364,0.0008407904,0.00074236275,0.00077278784,0.00045727473,0.001475152,0.0016540941,0.0013801733,0.009357654],"category_scores_gemma":[0.004621759,0.00039368667,0.0012448095,0.000710235,0.00061177654,0.003275502,0.0012208069,0.0020435601,0.0025297848],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038689442,0.00027641238,0.0021226536,0.00038556708,0.0003166813,0.00059979124,0.00049356517,0.52654666,0.011407036,0.2114589,0.01817525,0.22783063],"study_design_scores_gemma":[0.000010241914,0.000028104989,0.00018008382,0.000016266245,0.000028074664,0.000085487736,0.000029331224,0.95189023,0.0012794406,0.04454684,0.0018890243,0.000016902577],"about_ca_topic_score_codex":0.005433103,"about_ca_topic_score_gemma":0.006173786,"teacher_disagreement_score":0.009357654,"about_ca_system_score_codex":0.00096869835,"about_ca_system_score_gemma":0.0006817829,"threshold_uncertainty_score":0.03130448},"labels":[],"label_agreement":null},{"id":"W2172888184","doi":"10.48550/arxiv.1511.06361","title":"Order-Embeddings of Images and Language","year":2015,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":87,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Closed captioning; Hierarchy; Computer science; Natural language processing; Variety (cybernetics); Artificial intelligence; Image (mathematics); Logical consequence; Textual entailment; Order (exchange)","score_opus":0.04293762358184849,"score_gpt":0.21878959671675857,"score_spread":0.17585197313491008,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2172888184","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.042798407,0.0009101187,0.9438046,0.0009780112,0.00023717173,0.00016011702,0.0026857336,0.0018118228,0.006613933],"genre_scores_gemma":[0.57370734,0.0014113762,0.40665385,0.00044004506,0.00033891576,0.00044187397,0.0071124057,0.000544038,0.009350106],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99891317,0.0003391499,0.00008319283,0.00036068767,0.0002231389,0.00008061088],"domain_scores_gemma":[0.997502,0.0010147898,0.00031731272,0.0006905646,0.0003548603,0.000120481294],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008715401,0.00090693077,0.0005183077,0.0024658486,0.00045191072,0.001908385,0.0012045491,0.0013573233,0.0059285876],"category_scores_gemma":[0.007912377,0.0004529743,0.0011737708,0.0020794056,0.0012071508,0.0066665974,0.0018827352,0.0022171787,0.001906699],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048682984,0.00031143115,0.0044589634,0.0008915444,0.00013845133,0.00050175586,0.0016806436,0.046416376,0.017622963,0.4043639,0.021950044,0.5011771],"study_design_scores_gemma":[0.00003553895,0.00013245286,0.0016969147,0.00013887192,0.000058681733,0.00034754234,0.00035497473,0.4321733,0.0076362365,0.5378134,0.019559475,0.000052576084],"about_ca_topic_score_codex":0.002270081,"about_ca_topic_score_gemma":0.0039205956,"teacher_disagreement_score":0.0059285876,"about_ca_system_score_codex":0.00097173674,"about_ca_system_score_gemma":0.0008062169,"threshold_uncertainty_score":0.019833088},"labels":[],"label_agreement":null},{"id":"W2174492417","doi":"10.48550/arxiv.1511.05960","title":"ABC-CNN: An Attention Based Convolutional Neural Network for Visual Question Answering","year":2015,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":278,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Convolutional neural network; Question answering; Computer science; Artificial intelligence; Visual attention; Natural language processing; Pattern recognition (psychology); Psychology; Neuroscience; Cognition","score_opus":0.07866269093093313,"score_gpt":0.2518239291440776,"score_spread":0.17316123821314444,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2174492417","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09505685,0.0057290043,0.8430711,0.0018474847,0.000629882,0.000578391,0.0062786024,0.024203682,0.022604968],"genre_scores_gemma":[0.67781764,0.0016443344,0.2834142,0.0019710138,0.00017951378,0.00041371945,0.012936495,0.00044956603,0.021173498],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99973255,0.00003919641,0.000011292161,0.000110470166,0.00005443299,0.000051983432],"domain_scores_gemma":[0.9996561,0.000101336664,0.000028764429,0.00006896905,0.00011461352,0.000030222034],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00043287137,0.0012808382,0.0004830661,0.00072020595,0.0003079529,0.00067173026,0.0018554766,0.0013112338,0.005142367],"category_scores_gemma":[0.0013857696,0.00035080613,0.00073385186,0.0007255144,0.0004727645,0.0016450462,0.001068856,0.0014698026,0.0014948195],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004566159,0.00043627873,0.0050844937,0.0006280658,0.00023882402,0.00030939758,0.0002702696,0.10407624,0.05309794,0.018551659,0.078518234,0.738332],"study_design_scores_gemma":[0.000034710087,0.00011182008,0.0015495905,0.000044928434,0.00006191425,0.00011683003,0.000042409025,0.9514227,0.015052055,0.017102454,0.014436074,0.00002453229],"about_ca_topic_score_codex":0.02656222,"about_ca_topic_score_gemma":0.035793297,"teacher_disagreement_score":0.02656222,"about_ca_system_score_codex":0.0015633933,"about_ca_system_score_gemma":0.0010341214,"threshold_uncertainty_score":0.0528152},"labels":[],"label_agreement":null},{"id":"W2176212817","doi":"10.48550/arxiv.1511.06973","title":"Ask Me Anything: Free-form Visual Question Answering Based on Knowledge from External Sources","year":2015,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":47,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Question answering; Computer science; Image (mathematics); Ask price; Information retrieval; Knowledge base; Representation (politics); Questions and answers; Artificial intelligence; Knowledge representation and reasoning; Natural language processing","score_opus":0.04890827163354747,"score_gpt":0.23881712332946559,"score_spread":0.18990885169591812,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2176212817","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03804591,0.0016189013,0.9295266,0.001739133,0.00016732386,0.00059408386,0.0030048564,0.014433253,0.010869862],"genre_scores_gemma":[0.46978492,0.0005699994,0.5051621,0.0011832032,0.00024215296,0.0007619678,0.011216407,0.00054494734,0.010534281],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983972,0.00052317645,0.00007915237,0.0005939223,0.00028159536,0.00012501917],"domain_scores_gemma":[0.99667966,0.0020458181,0.00021093247,0.00048916356,0.00042457375,0.00014995586],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018411496,0.0017222393,0.00093683833,0.0019093606,0.00069785124,0.0020392023,0.0032120096,0.0029707034,0.008222589],"category_scores_gemma":[0.008908192,0.0005322894,0.0015741123,0.0009580211,0.001172144,0.0057082777,0.0029126965,0.0021323327,0.0024816096],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00090887106,0.0008421145,0.0048282095,0.0018299315,0.00029304757,0.00079948327,0.0027722374,0.052591186,0.038882274,0.039222337,0.062693246,0.7943371],"study_design_scores_gemma":[0.00012886945,0.00020907188,0.002232267,0.00014668143,0.00015887868,0.00039296533,0.0006359558,0.8469653,0.020501036,0.09819453,0.030343862,0.00009057436],"about_ca_topic_score_codex":0.010710691,"about_ca_topic_score_gemma":0.010879264,"teacher_disagreement_score":0.010710691,"about_ca_system_score_codex":0.0013845644,"about_ca_system_score_gemma":0.0011188069,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2190067570","doi":"10.1109/cvpr.2016.501","title":"MovieQA: Understanding Stories in Movies through Question-Answering","year":2016,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":55,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Karlsruhe House of Young Scientists; Deutsche Forschungsgemeinschaft","keywords":"Computer science; Question answering; Scripting language; Set (abstract data type); Benchmark (surveying); Comprehension; Semantics (computer science); Information retrieval; Natural language processing; Domain (mathematical analysis); CLIPS; Range (aeronautics); Artificial intelligence; Programming language","score_opus":0.06700293859533554,"score_gpt":0.34423839817472807,"score_spread":0.27723545957939255,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2190067570","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26147714,0.011528688,0.06955513,0.003919546,0.00092182885,0.0033242237,0.58543825,0.031541064,0.032294206],"genre_scores_gemma":[0.17164148,0.0009620519,0.119571045,0.0009484438,0.00024949576,0.0012166407,0.6981582,0.0005798649,0.0066727633],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9972608,0.0009945291,0.0002461016,0.0007592622,0.0005538707,0.00018537222],"domain_scores_gemma":[0.9938545,0.0036881154,0.00034631442,0.0009812003,0.0007144759,0.00041529085],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023192016,0.0025823002,0.001009445,0.003368802,0.0010545513,0.0020510452,0.0029277992,0.003326702,0.013761445],"category_scores_gemma":[0.014694566,0.00042325506,0.0013479353,0.0019718467,0.0006332244,0.005275941,0.0024854157,0.0025022798,0.0078088995],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002469207,0.003266492,0.024569986,0.0066761263,0.0006880354,0.00086285395,0.002601572,0.012367665,0.021306474,0.0068572173,0.6673708,0.25096348],"study_design_scores_gemma":[0.0013186961,0.0023213387,0.08776189,0.0009820097,0.00047068615,0.0026016126,0.004957375,0.28190792,0.037279163,0.026572093,0.55349535,0.0003319206],"about_ca_topic_score_codex":0.013445284,"about_ca_topic_score_gemma":0.023387793,"teacher_disagreement_score":0.013761445,"about_ca_system_score_codex":0.0014254969,"about_ca_system_score_gemma":0.0010740791,"threshold_uncertainty_score":0.0460366},"labels":[],"label_agreement":null},{"id":"W2467945436","doi":"10.1145/2932710","title":"Visualizing Natural Language Descriptions","year":2016,"lang":"en","type":"article","venue":"ACM Computing Surveys","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Exploit; Natural language user interface; Naturalness; Natural language programming; Natural language; Visualization; Simplicity; Natural language understanding; Language identification","score_opus":0.02202513094967928,"score_gpt":0.3171835668698234,"score_spread":0.2951584359201441,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2467945436","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008658764,0.0042069494,0.932764,0.002521672,0.00022770341,0.0003051714,0.0068131816,0.013288481,0.031214038],"genre_scores_gemma":[0.17339744,0.0065370295,0.78179055,0.000743329,0.00021445821,0.00053486036,0.015755124,0.0027862946,0.018240986],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9968221,0.0016321552,0.00026938407,0.00033678213,0.00083499425,0.00010448338],"domain_scores_gemma":[0.992334,0.005370815,0.00042919218,0.00080642744,0.00084671396,0.00021286335],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039309384,0.0013840563,0.0006730798,0.005137934,0.00077198696,0.007357223,0.0018567814,0.0012525164,0.018907268],"category_scores_gemma":[0.014888555,0.0006105772,0.0010394923,0.0034880042,0.0012438601,0.0103610065,0.0027832638,0.0016200105,0.004279269],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022286945,0.00010690121,0.0012748439,0.0021558853,0.00011189067,0.0007957139,0.009280929,0.008753263,0.00993893,0.6146554,0.069101684,0.2836017],"study_design_scores_gemma":[0.00007318326,0.000068381436,0.00047849174,0.0005441439,0.000061015136,0.000815558,0.0028555894,0.042903405,0.00720755,0.36055326,0.5843511,0.00008827367],"about_ca_topic_score_codex":0.0029336375,"about_ca_topic_score_gemma":0.004288165,"teacher_disagreement_score":0.018907268,"about_ca_system_score_codex":0.0013480021,"about_ca_system_score_gemma":0.0014071609,"threshold_uncertainty_score":0.06325114},"labels":[],"label_agreement":null},{"id":"W2524766041","doi":"10.5244/c.30.141","title":"Oracle Performance for Visual Captioning","year":2016,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Closed captioning; Computer science; Oracle; Task (project management); Artificial intelligence; Natural language; Process (computing); Natural language processing; Image (mathematics); Simplicity; Imperfect; Upper and lower bounds; Machine learning; Programming language; Mathematics; Linguistics","score_opus":0.021510436132043273,"score_gpt":0.3232858661331357,"score_spread":0.3017754300010924,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2524766041","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18617263,0.06298378,0.48476607,0.00814001,0.0050333757,0.0016420815,0.038682785,0.14101349,0.07156575],"genre_scores_gemma":[0.7241155,0.003957591,0.18501243,0.002261798,0.00082067144,0.0005400105,0.06670832,0.0026688532,0.013914792],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99424475,0.0018909607,0.0003922154,0.0016830835,0.0011665402,0.00062250614],"domain_scores_gemma":[0.984539,0.0093785655,0.0004936888,0.0035659128,0.0013972655,0.00062564475],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006630834,0.0026474632,0.002720376,0.0024364162,0.0009782987,0.0042309407,0.0037383083,0.0047207233,0.018729022],"category_scores_gemma":[0.0340186,0.000548157,0.0015290253,0.0018905011,0.0015649974,0.005466199,0.0041868757,0.003860016,0.010425164],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003163995,0.0005876808,0.003244225,0.0022787424,0.00030057706,0.00038527377,0.00016109027,0.19577274,0.012004001,0.008979069,0.113149635,0.6599731],"study_design_scores_gemma":[0.00016960435,0.0005483319,0.002310159,0.000205812,0.00008380937,0.0005381078,0.00014634438,0.94778466,0.017087217,0.017340425,0.013677271,0.000108150714],"about_ca_topic_score_codex":0.012586034,"about_ca_topic_score_gemma":0.01036127,"teacher_disagreement_score":0.018729022,"about_ca_system_score_codex":0.0039567975,"about_ca_system_score_gemma":0.0020488445,"threshold_uncertainty_score":0.06265485},"labels":[],"label_agreement":null},{"id":"W2557264465","doi":"10.1109/cvpr.2017.339","title":"Hierarchical Boundary-Aware Neural Encoder for Video Captioning","year":2017,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":141,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Ministero dell’Istruzione, dell’Università e della Ricerca","keywords":"Closed captioning; ENCODE; Encoding (memory); Leverage (statistics); Encoder; Recurrent neural network; Artificial neural network; Scheme (mathematics); Annotation","score_opus":0.02768488918146487,"score_gpt":0.32825486747151345,"score_spread":0.30056997829004856,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2557264465","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03686822,0.0016488195,0.9461313,0.0003247048,0.00021058266,0.00012621886,0.0015237648,0.009136291,0.004030154],"genre_scores_gemma":[0.52268374,0.0010500845,0.46041226,0.00039092335,0.0001738798,0.0002617057,0.0058066435,0.00049587036,0.008724906],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99963486,0.00009511411,0.00002025071,0.00011283173,0.00008850514,0.000048362876],"domain_scores_gemma":[0.9993299,0.00025674223,0.000073504096,0.00012580928,0.00017981847,0.000034241733],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005466664,0.0008818221,0.0007151489,0.00075147214,0.00026557443,0.0005359583,0.0012718,0.00087403867,0.0035091315],"category_scores_gemma":[0.0030476279,0.000326806,0.0005131779,0.000810114,0.00038323406,0.0017817772,0.00085757376,0.001378148,0.0015885805],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044109244,0.00019344673,0.0007029035,0.00035996072,0.00007057424,0.0003180968,0.00020743735,0.119616844,0.062030725,0.010494883,0.021347038,0.7842171],"study_design_scores_gemma":[0.0000132675905,0.000055934655,0.00024709298,0.000019164345,0.000018933742,0.0000661685,0.000024910729,0.97502035,0.017356735,0.0047926493,0.0023694606,0.000015324298],"about_ca_topic_score_codex":0.005293994,"about_ca_topic_score_gemma":0.007906853,"teacher_disagreement_score":0.005293994,"about_ca_system_score_codex":0.0007460866,"about_ca_system_score_gemma":0.0005605929,"threshold_uncertainty_score":0.011739194},"labels":[],"label_agreement":null},{"id":"W2560645892","doi":"10.1109/iccv.2017.140","title":"Areas of Attention for Image Captioning","year":2017,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":218,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"Agence Nationale de la Recherche","keywords":"Closed captioning; Computer science; Pairwise comparison; Transformer; Artificial intelligence; Convolutional neural network; Image (mathematics); Language model; Pattern recognition (psychology); Computer vision","score_opus":0.030481745100812682,"score_gpt":0.3389131894072359,"score_spread":0.3084314443064232,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2560645892","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049684197,0.0035294387,0.90484285,0.0008990293,0.00045697318,0.00038604427,0.0021857012,0.026125342,0.011890389],"genre_scores_gemma":[0.61402506,0.0014163374,0.36209697,0.00076029374,0.0003740201,0.00044155522,0.006080111,0.0015869344,0.013218673],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990984,0.00023331704,0.000039403803,0.0003570968,0.00015065073,0.000121061064],"domain_scores_gemma":[0.9986603,0.00047402526,0.00012778907,0.0003908764,0.00026953977,0.000077523946],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011072546,0.0016907203,0.0007889174,0.0017146791,0.00051192037,0.0015607532,0.0021692798,0.001540958,0.006123016],"category_scores_gemma":[0.0039529693,0.0005401542,0.0014555238,0.0011898527,0.00096902926,0.0033347497,0.0021230641,0.0022831876,0.0027084127],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00073710555,0.00024926415,0.0028427127,0.00063137064,0.00019882074,0.00029159442,0.00045237492,0.11432681,0.04423166,0.01877627,0.03934645,0.7779157],"study_design_scores_gemma":[0.00003200288,0.00012946074,0.0013206152,0.00004236658,0.00007064534,0.00020972049,0.000067077686,0.9267137,0.032743588,0.020476758,0.018160453,0.0000335176],"about_ca_topic_score_codex":0.005393265,"about_ca_topic_score_gemma":0.0074132495,"teacher_disagreement_score":0.006123016,"about_ca_system_score_codex":0.0013252951,"about_ca_system_score_gemma":0.0009001623,"threshold_uncertainty_score":0.020483494},"labels":[],"label_agreement":null},{"id":"W2562029256","doi":"10.1109/crv.2016.39","title":"Learning Neural Networks with Ranking-Based Losses for Action Retrieval","year":2016,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Softmax function; Hinge loss; Artificial neural network; Computer science; Ranking (information retrieval); Artificial intelligence; Machine learning; Pattern recognition (psychology); Function (biology); Support vector machine","score_opus":0.019387711974587274,"score_gpt":0.28219914199966795,"score_spread":0.2628114300250807,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2562029256","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.058344834,0.0031833148,0.93087614,0.000787627,0.0001403261,0.00013156144,0.00018462571,0.0012517563,0.0050998065],"genre_scores_gemma":[0.8186311,0.0014744565,0.16583444,0.0005876104,0.00027111624,0.00025735536,0.00068368885,0.00015609586,0.012104075],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99944717,0.00014711627,0.000039893956,0.00014230727,0.00014329885,0.000080101294],"domain_scores_gemma":[0.9992021,0.00041606024,0.00010240245,0.00007292916,0.00017311482,0.000033459783],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015824874,0.0011456759,0.0014811059,0.00085104926,0.0003671089,0.0009778044,0.0014065224,0.0018870932,0.0025261166],"category_scores_gemma":[0.0040208013,0.00033453683,0.00062736624,0.001225547,0.00057801517,0.0022603183,0.0007005095,0.0015525277,0.00091317826],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000252308,0.00029323224,0.00082551996,0.00016898208,0.00008630521,0.0001179964,0.000041495518,0.7651056,0.005650774,0.0055875937,0.003655322,0.21821487],"study_design_scores_gemma":[0.0000040315595,0.000035113004,0.000082465885,0.0000037123648,0.00000636414,0.000010654443,0.0000038789,0.99711645,0.00064748403,0.0019084931,0.00017743591,0.000003845448],"about_ca_topic_score_codex":0.0047830106,"about_ca_topic_score_gemma":0.0040317043,"teacher_disagreement_score":0.0047830106,"about_ca_system_score_codex":0.0012773715,"about_ca_system_score_gemma":0.000674852,"threshold_uncertainty_score":0.009510338},"labels":[],"label_agreement":null},{"id":"W2575068822","doi":"","title":"A Scalable Unsupervised Deep Multimodal Learning System.","year":2016,"lang":"en","type":"article","venue":"The Florida AI Research Society","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Acadia University","funders":"","keywords":"Computer science; Deep learning; Artificial intelligence; Scalability; Unsupervised learning; Machine learning; Database","score_opus":0.03201505997780801,"score_gpt":0.33672308778766696,"score_spread":0.30470802780985895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2575068822","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04936288,0.0023269816,0.88364613,0.0009887059,0.00063052966,0.00036215008,0.0047709625,0.043373898,0.014537749],"genre_scores_gemma":[0.47606698,0.00078151026,0.47960916,0.0011163743,0.00021853243,0.00075213605,0.0099483915,0.0011086314,0.030398337],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99973947,0.000053480046,0.000010644965,0.00008632142,0.00006035202,0.000049760303],"domain_scores_gemma":[0.99976355,0.00005481517,0.0000128536,0.000050188337,0.00008965232,0.000028966724],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005144184,0.0009046471,0.0006876847,0.0005354446,0.00043318694,0.00061651017,0.0016693217,0.00094138854,0.010176409],"category_scores_gemma":[0.0015366728,0.0004122466,0.0005655182,0.0005036355,0.00026381572,0.0013884221,0.0022666103,0.0015513026,0.004417724],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006424661,0.0004574355,0.0019062021,0.00020369794,0.00025788762,0.00020152572,0.00007588247,0.04017216,0.03226695,0.0050177886,0.079756126,0.8390419],"study_design_scores_gemma":[0.00006578727,0.00012247378,0.00081031676,0.000034683002,0.000051702813,0.00012872124,0.00003470633,0.9681576,0.012781799,0.00874926,0.009033872,0.000029092042],"about_ca_topic_score_codex":0.00650803,"about_ca_topic_score_gemma":0.015126678,"teacher_disagreement_score":0.010176409,"about_ca_system_score_codex":0.00061598327,"about_ca_system_score_gemma":0.0009205802,"threshold_uncertainty_score":0.03404349},"labels":[],"label_agreement":null},{"id":"W2587335008","doi":"10.29173/cais427","title":"Task-Based Representation of Moving Images","year":2013,"lang":"en","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Task (project management); Representation (politics); Computer science; Construct (python library); Congruence (geometry); Artificial intelligence; Computer vision; Image (mathematics); Psychology","score_opus":0.01778663194423156,"score_gpt":0.2627378118163914,"score_spread":0.24495117987215984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2587335008","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6149787,0.00072571525,0.36209646,0.0005383117,0.000119712524,0.00048298767,0.0012018847,0.0008026081,0.019053647],"genre_scores_gemma":[0.96003443,0.0001350859,0.037625454,0.000041237352,0.000026090998,0.00009982149,0.000711383,0.00009126877,0.0012353534],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99887794,0.00037849197,0.000080913494,0.00034413332,0.00021352782,0.00010499903],"domain_scores_gemma":[0.9941538,0.0020130286,0.0013408908,0.0011696662,0.001090697,0.0002320149],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017378831,0.00037752022,0.00035071652,0.0009322704,0.0003536221,0.0019246276,0.00091495796,0.0005301547,0.0027397536],"category_scores_gemma":[0.023068022,0.00023133788,0.0005136947,0.00081723946,0.00059449044,0.0037673262,0.0014878659,0.0006870885,0.0005077324],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0045112707,0.00054688245,0.04336215,0.0014686604,0.0003440632,0.0005141156,0.011212309,0.055548843,0.25875086,0.05974078,0.0064171953,0.557583],"study_design_scores_gemma":[0.00026386438,0.0030120653,0.20090714,0.00042482416,0.00049305666,0.0011582429,0.0077234646,0.62404907,0.066700354,0.071761,0.023156688,0.00035016073],"about_ca_topic_score_codex":0.0038620937,"about_ca_topic_score_gemma":0.0015459108,"teacher_disagreement_score":0.0038620937,"about_ca_system_score_codex":0.0006187202,"about_ca_system_score_gemma":0.00052809645,"threshold_uncertainty_score":0.009190857},"labels":[],"label_agreement":null},{"id":"W2599940792","doi":"10.24963/ijcai.2017/385","title":"End-to-end optimization of goal-driven and visually grounded dialogue systems","year":2017,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":89,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Agence Nationale de la Recherche; Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Canadian Institute for Advanced Research","keywords":"Computer science; Task (project management); Utterance; Artificial intelligence; Reinforcement learning; Sequence (biology); Object (grammar); Human–computer interaction; Engineering","score_opus":0.017686626201186408,"score_gpt":0.29341083613172203,"score_spread":0.27572420993053565,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2599940792","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06235531,0.00030678717,0.9313652,0.00032339187,0.000054321947,0.00016910741,0.000093398536,0.001728103,0.0036044007],"genre_scores_gemma":[0.8633749,0.00007855945,0.13297254,0.00015204694,0.00002066414,0.00029418862,0.00017510113,0.00025989514,0.0026721957],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992028,0.00034311495,0.00003179239,0.00019572226,0.000106341126,0.00012022996],"domain_scores_gemma":[0.99822086,0.0012183064,0.00011352344,0.00009052812,0.00021755951,0.00013924792],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020121646,0.0014626142,0.0010770289,0.00041205154,0.0005077871,0.0010239771,0.0011479028,0.001684338,0.0029865366],"category_scores_gemma":[0.0050305664,0.0006681335,0.0005389066,0.00019995627,0.0010550709,0.00096685556,0.0018415537,0.0016662364,0.00059833593],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016142691,0.00008772604,0.00032523074,0.00007179348,0.000024571638,0.000077063494,0.0001349516,0.962439,0.0033378282,0.0026727314,0.0007562634,0.029911377],"study_design_scores_gemma":[0.000011395948,0.000024215255,0.000034040386,0.0000034725922,0.0000024098927,0.000005501615,0.000010776438,0.99774903,0.00058707374,0.0014376314,0.00013188284,0.0000024782753],"about_ca_topic_score_codex":0.0037939798,"about_ca_topic_score_gemma":0.0037669651,"teacher_disagreement_score":0.0037939798,"about_ca_system_score_codex":0.0013254938,"about_ca_system_score_gemma":0.001754693,"threshold_uncertainty_score":0.010641456},"labels":[],"label_agreement":null},{"id":"W2608022654","doi":"10.1109/icpr.2016.7900081","title":"Automatic video description generation via LSTM with joint two-stream encoding","year":2016,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Encoding (memory); Artificial intelligence; Convolutional neural network; Closed captioning; Recurrent neural network; RGB color model; Decoding methods; Deep learning; Component (thermodynamics); Feature extraction; Pattern recognition (psychology); Artificial neural network; Image (mathematics); Algorithm","score_opus":0.02855375016983549,"score_gpt":0.2545483060394659,"score_spread":0.22599455586963038,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2608022654","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01900011,0.00075630326,0.9613233,0.00033410024,0.00024293066,0.00020159876,0.0010794005,0.013625194,0.0034370702],"genre_scores_gemma":[0.3416218,0.0008789098,0.63874197,0.00042567708,0.00016660614,0.0003599626,0.006339664,0.0006592091,0.010806209],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996382,0.00005658678,0.000024477285,0.000120809564,0.00011115402,0.000048723065],"domain_scores_gemma":[0.9995059,0.00016570624,0.000043031225,0.00009829424,0.00015231226,0.000034808327],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00053093396,0.0013480516,0.00072833593,0.0009885503,0.00026428647,0.0008099422,0.0019219009,0.00094325654,0.0042304583],"category_scores_gemma":[0.0020043754,0.00039533334,0.0006997842,0.0010862863,0.0004110895,0.0024614448,0.0010903566,0.0014824437,0.0018948591],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043500736,0.00022722232,0.00067565014,0.0002928915,0.00009298756,0.00038318746,0.00012243108,0.089741,0.041645553,0.0074213715,0.018540474,0.8404222],"study_design_scores_gemma":[0.000027465765,0.000059073805,0.000177293,0.000014805128,0.000024050103,0.0001004677,0.000033611766,0.9666449,0.024343465,0.0045128944,0.0040454925,0.0000164566],"about_ca_topic_score_codex":0.0064495075,"about_ca_topic_score_gemma":0.0069219144,"teacher_disagreement_score":0.0064495075,"about_ca_system_score_codex":0.00096626347,"about_ca_system_score_gemma":0.00083347247,"threshold_uncertainty_score":0.014152348},"labels":[],"label_agreement":null},{"id":"W2617136920","doi":"10.24963/ijcai.2017/178","title":"How a General-Purpose Commonsense Ontology can Improve Performance of Learning-Based Image Retrieval","year":2017,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Comisión Nacional de Investigación Científica y Tecnológica","keywords":"Commonsense knowledge; Computer science; Ontology; Exploit; Commonsense reasoning; Artificial intelligence; Information retrieval; Natural language processing; Question answering; Testbed; Sentence; Visualization; Knowledge retrieval; Benchmark (surveying); Knowledge representation and reasoning; Knowledge extraction; World Wide Web","score_opus":0.014062946221605574,"score_gpt":0.269845979589982,"score_spread":0.25578303336837643,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2617136920","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.38708878,0.0072416035,0.53876376,0.0028984733,0.00073323894,0.0007828316,0.0035910932,0.031631563,0.027268607],"genre_scores_gemma":[0.693978,0.0012447407,0.29119235,0.0007474875,0.000099087636,0.000121800964,0.008331842,0.0004517073,0.0038328608],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99833316,0.00033562988,0.00018349235,0.00043218216,0.0005154965,0.00020001411],"domain_scores_gemma":[0.99783957,0.00074546895,0.00010676115,0.00080933596,0.00041017946,0.0000885765],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030391512,0.0013745028,0.0011172029,0.0024655906,0.0010807861,0.0018766893,0.001880882,0.0017969923,0.0038732747],"category_scores_gemma":[0.008625686,0.0003156592,0.001117977,0.0021847747,0.0008267353,0.007678316,0.0020986493,0.0016467661,0.0020482752],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005801807,0.0006623131,0.0054407357,0.00069574034,0.00024568342,0.00021022657,0.0002820277,0.042104002,0.03510701,0.009381592,0.028763456,0.8765271],"study_design_scores_gemma":[0.0001568907,0.00057817105,0.004442497,0.00012140236,0.00032961988,0.000625949,0.00070686213,0.8471454,0.06802908,0.038341578,0.039408416,0.00011409243],"about_ca_topic_score_codex":0.016779259,"about_ca_topic_score_gemma":0.020561896,"teacher_disagreement_score":0.016779259,"about_ca_system_score_codex":0.0016217614,"about_ca_system_score_gemma":0.0017506067,"threshold_uncertainty_score":0.033363163},"labels":[],"label_agreement":null},{"id":"W2734498959","doi":"10.48550/arxiv.1709.07871","title":"FiLM: Visual Reasoning with a General Conditioning Layer","year":2017,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":189,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Affine transformation; Computer science; Benchmark (surveying); Feature (linguistics); Artificial intelligence; Transformation (genetics); Computation; Simple (philosophy); Artificial neural network; Layer (electronics); Image (mathematics); Task (project management); Process (computing); Pattern recognition (psychology); Machine learning; Algorithm; Mathematics; Engineering","score_opus":0.05285417320652501,"score_gpt":0.23371162585806823,"score_spread":0.18085745265154324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2734498959","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020863099,0.0004443693,0.9575563,0.0006042538,0.0001620012,0.00018520997,0.00055617886,0.011037438,0.008591094],"genre_scores_gemma":[0.5194729,0.00037453763,0.46533337,0.0009391079,0.00010949623,0.00029409796,0.0011659521,0.0010005927,0.011309998],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99969256,0.000054065513,0.000016839926,0.00011393025,0.00007843997,0.000044199427],"domain_scores_gemma":[0.9994597,0.00016772663,0.000052699466,0.0002166441,0.000060849765,0.000042416872],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006610155,0.00102828,0.00045006047,0.00032876004,0.000274618,0.0012879474,0.0018653008,0.0011019086,0.013546911],"category_scores_gemma":[0.0029769256,0.0004916331,0.0006196535,0.00026225232,0.00096675334,0.003394955,0.0020524159,0.002115168,0.0022189242],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010129533,0.00044032885,0.0015661212,0.0007074808,0.00023463229,0.00027928795,0.00032260246,0.13918325,0.11237237,0.08695943,0.03264521,0.6242763],"study_design_scores_gemma":[0.00009902248,0.00018808861,0.0004797252,0.000049399012,0.000058071528,0.00011880056,0.000032364762,0.8702818,0.059226986,0.056260433,0.013175697,0.000029605491],"about_ca_topic_score_codex":0.0020514561,"about_ca_topic_score_gemma":0.0029731279,"teacher_disagreement_score":0.013546911,"about_ca_system_score_codex":0.0006640726,"about_ca_system_score_gemma":0.0006375542,"threshold_uncertainty_score":0.04531896},"labels":[],"label_agreement":null},{"id":"W2741237521","doi":"10.24963/ijcai.2017/479","title":"Global-residual and Local-boundary Refinement Networks for Rectifying Scene Parsing Predictions","year":2017,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Parsing; Residual; Computer science; Boundary (topology); Artificial intelligence; Convolutional neural network; Object (grammar); Cascade; Iterative refinement; Machine learning; Data mining; Pattern recognition (psychology); Algorithm; Mathematics","score_opus":0.026982535933759544,"score_gpt":0.32188678243282126,"score_spread":0.2949042464990617,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2741237521","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040388335,0.000648517,0.9461803,0.00031174836,0.00009737232,0.000082792954,0.00045432587,0.008347913,0.0034887034],"genre_scores_gemma":[0.38932857,0.00039040414,0.59733266,0.0004208573,0.00007333414,0.00014702459,0.0026394546,0.0009522802,0.008715507],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993987,0.000083667954,0.0000236042,0.00030008773,0.00012002622,0.0000739683],"domain_scores_gemma":[0.9993461,0.00016540283,0.000073888965,0.00017131043,0.00020526336,0.000038065842],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001102994,0.0015060223,0.00070176966,0.0010352209,0.0005490876,0.00065319863,0.0025234157,0.001405322,0.0037615895],"category_scores_gemma":[0.002909275,0.00061921956,0.00089144096,0.0008364123,0.0006664869,0.0024161292,0.0014354966,0.002106516,0.0016073702],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027762348,0.00015586325,0.0024279254,0.00015654806,0.00008680129,0.00019896656,0.00025279835,0.27714965,0.03069088,0.010720777,0.013596941,0.6642852],"study_design_scores_gemma":[0.000012287322,0.00003048555,0.0005998577,0.000013682684,0.000027538666,0.000037954353,0.000028101778,0.98301274,0.009041995,0.004883639,0.00229484,0.000016851798],"about_ca_topic_score_codex":0.012552477,"about_ca_topic_score_gemma":0.025378115,"teacher_disagreement_score":0.012552477,"about_ca_system_score_codex":0.00082239386,"about_ca_system_score_gemma":0.00134487,"threshold_uncertainty_score":0.02495879},"labels":[],"label_agreement":null},{"id":"W2748195089","doi":"10.48550/arxiv.1708.04320","title":"Situation Recognition with Graph Neural Networks","year":2017,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Nvidia","keywords":"Computer science; Salient; Graph; Artificial intelligence; Artificial neural network; Action (physics); Verb; Noun; Task (project management); Machine learning; Natural language processing; Pattern recognition (psychology); Theoretical computer science","score_opus":0.07201039835022935,"score_gpt":0.20474852014659253,"score_spread":0.13273812179636318,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2748195089","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1275025,0.0013015873,0.8556032,0.0014367618,0.00021087906,0.00012336027,0.0008953142,0.0045021763,0.008424185],"genre_scores_gemma":[0.82569385,0.0005011963,0.16890237,0.0003027851,0.000077146164,0.000077587676,0.0012804355,0.00013299199,0.0030316277],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999637,0.00008840249,0.000015844964,0.00016129711,0.00005352334,0.000043968816],"domain_scores_gemma":[0.99936,0.00028869984,0.0001123544,0.000098480865,0.00009492189,0.000045524044],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004033314,0.001048278,0.00045124092,0.0015275317,0.00049062865,0.0010342224,0.0011614083,0.001218072,0.0021037732],"category_scores_gemma":[0.002336823,0.0005001323,0.0009200366,0.001209793,0.00063766964,0.0028397199,0.00082914776,0.0013669945,0.0006767607],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028279878,0.00028303458,0.0062938253,0.00020119005,0.00018942426,0.00036744672,0.00028354547,0.6321286,0.0106413355,0.024175918,0.008488323,0.3166646],"study_design_scores_gemma":[0.0000052965,0.000018327664,0.00053127744,0.000008991418,0.000014652361,0.00003276564,0.00002597058,0.96729034,0.0014131801,0.029877685,0.00077232515,0.000009145816],"about_ca_topic_score_codex":0.011300785,"about_ca_topic_score_gemma":0.015147585,"teacher_disagreement_score":0.011300785,"about_ca_system_score_codex":0.0011490346,"about_ca_system_score_gemma":0.00049871195,"threshold_uncertainty_score":0.022469997},"labels":[],"label_agreement":null},{"id":"W2760103357","doi":"10.1609/aaai.v32i1.11671","title":"FiLM: Visual Reasoning with a General Conditioning Layer","year":2018,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1633,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Institute for Advanced Research; Université de Montréal","funders":"Fonds de recherche du Québec – Nature et technologies; CHIST-ERA; Nvidia","keywords":"Affine transformation; Computer science; Benchmark (surveying); Artificial intelligence; Feature (linguistics); Computation; Transformation (genetics); Artificial neural network; Simple (philosophy); Layer (electronics); Visual reasoning; Image (mathematics); Task (project management); Process (computing); Pattern recognition (psychology); Machine learning; Algorithm; Mathematics","score_opus":0.04609872896526987,"score_gpt":0.3260767922855972,"score_spread":0.2799780633203274,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2760103357","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020138491,0.00042676314,0.95615685,0.0005475299,0.00017197907,0.00019983244,0.000560865,0.013050412,0.00874723],"genre_scores_gemma":[0.49523762,0.00036806279,0.48895645,0.0009619433,0.00010655314,0.000303953,0.001202365,0.0011166387,0.011746417],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996909,0.000049867143,0.000017917586,0.00011195631,0.000083538755,0.000045905563],"domain_scores_gemma":[0.99943835,0.00017431773,0.000054488894,0.00022342255,0.00006621449,0.000043148724],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00063973083,0.0010683571,0.00045021923,0.0003313238,0.00027467744,0.0013059567,0.0019269614,0.0010763407,0.013792588],"category_scores_gemma":[0.0028415504,0.000518426,0.000651501,0.00024756807,0.00093391695,0.0034541718,0.0019593542,0.0021373995,0.0022813126],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010890776,0.0004489443,0.0014442003,0.00075225945,0.0002412783,0.00028349823,0.00032042546,0.12982945,0.12522279,0.080016285,0.033183727,0.62716794],"study_design_scores_gemma":[0.00011447763,0.00021771948,0.0004922131,0.00005354484,0.00006839664,0.00013097342,0.000033424203,0.8590227,0.073099315,0.051239736,0.015493973,0.00003364491],"about_ca_topic_score_codex":0.0022113086,"about_ca_topic_score_gemma":0.0031121224,"teacher_disagreement_score":0.013792588,"about_ca_system_score_codex":0.00067819597,"about_ca_system_score_gemma":0.0006646126,"threshold_uncertainty_score":0.04614073},"labels":[],"label_agreement":null},{"id":"W2766520430","doi":"10.1145/3123266.3123327","title":"Catching the Temporal Regions-of-Interest for Video Captioning","year":2017,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":80,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"National Natural Science Foundation of China","keywords":"Closed captioning; Computer science; Focus (optics); Representation (politics); Frame (networking); Artificial intelligence; Region of interest; Image (mathematics); Telecommunications","score_opus":0.1113495865189584,"score_gpt":0.3552552529180734,"score_spread":0.243905666399115,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2766520430","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037678767,0.0061724377,0.9305654,0.0006854079,0.00053718156,0.00058529567,0.002664119,0.015356035,0.0057553146],"genre_scores_gemma":[0.39063686,0.00440103,0.57999116,0.00081002514,0.00089074305,0.000594984,0.011512707,0.00206978,0.009092587],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99871457,0.00035090832,0.00007292897,0.00046493748,0.00025982712,0.00013687595],"domain_scores_gemma":[0.9968934,0.0012050548,0.00031637875,0.0006200693,0.0008201913,0.00014485135],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015847945,0.0023465704,0.001378312,0.002201023,0.00058956636,0.0012798079,0.0017743221,0.0017573085,0.0028356265],"category_scores_gemma":[0.0076083317,0.00051764265,0.0011791257,0.001898861,0.00075792824,0.002970555,0.0016078454,0.002328684,0.0026389807],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008443508,0.00025326223,0.001516542,0.0011197359,0.0001723099,0.000495588,0.00052917696,0.0494278,0.110576116,0.0039031606,0.04275092,0.788411],"study_design_scores_gemma":[0.000056930094,0.00029628503,0.0025134524,0.00012287743,0.00019450295,0.00071763655,0.00021095526,0.8969286,0.06982582,0.00583574,0.023214666,0.0000824927],"about_ca_topic_score_codex":0.0066404776,"about_ca_topic_score_gemma":0.009345894,"teacher_disagreement_score":0.0066404776,"about_ca_system_score_codex":0.0009263985,"about_ca_system_score_gemma":0.0008276287,"threshold_uncertainty_score":0.013203621},"labels":[],"label_agreement":null},{"id":"W2767290858","doi":"10.1109/msp.2017.2738401","title":"Deep Multimodal Learning: A Survey on Recent Advances and Trends","year":2017,"lang":"en","type":"article","venue":"IEEE Signal Processing Magazine","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1067,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Deep learning; Computer science; Artificial intelligence; Modalities; Multimodal learning; Field (mathematics); Machine learning; Data science","score_opus":0.027873629156456387,"score_gpt":0.3174627048172625,"score_spread":0.28958907566080616,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2767290858","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00519095,0.90817803,0.06436006,0.0051622353,0.00069226755,0.00006301374,0.00038990183,0.00042949023,0.015534054],"genre_scores_gemma":[0.037565984,0.919416,0.032898325,0.002054714,0.0010988687,0.00009333131,0.0008143654,0.00013971899,0.005918615],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996333,0.000071466515,0.000040546136,0.00007475433,0.00014820442,0.000031708874],"domain_scores_gemma":[0.99873704,0.0007360709,0.00006855372,0.00004972942,0.0003387785,0.00006991217],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001300369,0.00097379886,0.0008777864,0.0019624394,0.00024622344,0.0013002651,0.0013252344,0.0010097697,0.006409968],"category_scores_gemma":[0.0036020246,0.0004462422,0.0005408275,0.0026545878,0.00048111353,0.0031003747,0.0012469722,0.0017774242,0.0020551542],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000081264334,0.00006801814,0.0012532208,0.0025035297,0.00006493377,0.00006856299,0.0000644957,0.0046313666,0.0008894861,0.012024015,0.026000421,0.95235074],"study_design_scores_gemma":[0.00003796822,0.000295869,0.0033150865,0.005118185,0.0002702497,0.001077669,0.00034186244,0.04961083,0.005154359,0.04942865,0.88522583,0.00012331727],"about_ca_topic_score_codex":0.0020655605,"about_ca_topic_score_gemma":0.0028608907,"teacher_disagreement_score":0.006409968,"about_ca_system_score_codex":0.0008791328,"about_ca_system_score_gemma":0.0010469228,"threshold_uncertainty_score":0.021443486},"labels":[],"label_agreement":null},{"id":"W2775835662","doi":"","title":"IC Decamouaging:: Reverse Engineering Camouflaged ICs within Minutes.","year":2015,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reverse engineering; Engineering; Forensic engineering; Computer science; Operating system","score_opus":0.02502063523506517,"score_gpt":0.24901010794289805,"score_spread":0.22398947270783287,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2775835662","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1530547,0.0054851687,0.59083575,0.0023966907,0.0047572516,0.0009858287,0.0017755075,0.0412941,0.19941506],"genre_scores_gemma":[0.60936284,0.0017862924,0.2643942,0.0016444502,0.00028480467,0.0002545162,0.0014981604,0.0048189852,0.11595575],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9994716,0.000041538322,0.000016923115,0.00007863877,0.0002953681,0.00009599171],"domain_scores_gemma":[0.9990429,0.00016893764,0.000056448553,0.00044074998,0.00024693483,0.000044063923],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004863618,0.0013040719,0.00047205255,0.0008436321,0.0006759691,0.0014033822,0.0012891993,0.001175761,0.029980335],"category_scores_gemma":[0.0023592273,0.00038689256,0.00043998237,0.0004189594,0.0007659037,0.0014236736,0.0018888788,0.0011547155,0.008122319],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063217635,0.00015658919,0.0017406467,0.0007048111,0.00008054835,0.0014432712,0.0007297782,0.008067212,0.24561258,0.017144809,0.067999795,0.6556878],"study_design_scores_gemma":[0.000087450346,0.000660528,0.0021561526,0.00025180585,0.00011137045,0.0028096414,0.0006624047,0.04522907,0.5851252,0.009706445,0.3530856,0.0001143144],"about_ca_topic_score_codex":0.0010673363,"about_ca_topic_score_gemma":0.0028760456,"teacher_disagreement_score":0.029980335,"about_ca_system_score_codex":0.0003942718,"about_ca_system_score_gemma":0.000717593,"threshold_uncertainty_score":0.10029417},"labels":[],"label_agreement":null},{"id":"W2788713610","doi":"10.1109/wacv.2019.00048","title":"Joint Event Detection and Description in Continuous Video Streams","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Closed captioning; Computer science; Event (particle physics); Context (archaeology); Pooling; Feature (linguistics); Joint (building); Artificial intelligence; Task (project management); Feature extraction; Encoding (memory); Image (mathematics)","score_opus":0.018053724091964162,"score_gpt":0.2600056807491931,"score_spread":0.24195195665722896,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2788713610","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10718234,0.0037367416,0.8394915,0.0010126758,0.00053700164,0.00085692975,0.014073368,0.026831552,0.0062779007],"genre_scores_gemma":[0.43633118,0.001600244,0.4795649,0.0004829448,0.00045012144,0.00053515984,0.072239555,0.0008261783,0.007969627],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99837697,0.00032921418,0.0000823787,0.0006438559,0.00039799203,0.00016955174],"domain_scores_gemma":[0.9968953,0.0014134774,0.00029817704,0.000701197,0.00054258096,0.00014923282],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019441812,0.0030560321,0.0015720602,0.0027735087,0.0006451596,0.0022398215,0.0025647043,0.0021628765,0.0035961785],"category_scores_gemma":[0.008122833,0.0006119926,0.0011052075,0.0028928926,0.00076086127,0.0047694813,0.0020145646,0.002530204,0.001909648],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020327533,0.00051268016,0.004163706,0.00095562835,0.00026703477,0.00080621033,0.00033625527,0.16143288,0.024643255,0.0056807976,0.05160366,0.74756515],"study_design_scores_gemma":[0.000056657274,0.0001711747,0.0015280457,0.00004558092,0.000049352122,0.00024075426,0.000121660116,0.96163195,0.01881835,0.007302817,0.00999722,0.000036363366],"about_ca_topic_score_codex":0.008379043,"about_ca_topic_score_gemma":0.010761187,"teacher_disagreement_score":0.008379043,"about_ca_system_score_codex":0.0015271127,"about_ca_system_score_gemma":0.0010846019,"threshold_uncertainty_score":0.016660571},"labels":[],"label_agreement":null},{"id":"W2789177853","doi":"10.1609/aaai.v32i1.12274","title":"Visual Relationship Detection With Deep Structural Ranking","year":2018,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":96,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"China Scholarship Council; Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China; Canada Research Chairs","keywords":"Ranking (information retrieval); Computer science; Artificial intelligence; Complement (music); Representation (politics); Relevance (law); Object (grammar); Machine learning; Pattern recognition (psychology); Function (biology); Image (mathematics)","score_opus":0.047743937746706655,"score_gpt":0.31674477158315684,"score_spread":0.2690008338364502,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2789177853","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.070751205,0.0024006094,0.909874,0.000328491,0.0001357364,0.00023950182,0.0012966845,0.009860622,0.005113162],"genre_scores_gemma":[0.6267779,0.0009357023,0.35606596,0.0004190536,0.00018610206,0.0002171152,0.0057613864,0.00055424275,0.009082506],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987325,0.00018735378,0.00005841866,0.00042750133,0.00039644225,0.00019771034],"domain_scores_gemma":[0.99875,0.00031601297,0.00023975951,0.00030437598,0.00027573342,0.000114044706],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008348082,0.0018960246,0.0014006951,0.0040160986,0.0006654232,0.0015422778,0.0031787206,0.0016588925,0.0037774674],"category_scores_gemma":[0.0028350926,0.0005941763,0.0012113451,0.0025759167,0.00048433078,0.0031531383,0.0020151553,0.0015862173,0.0022136918],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044474335,0.00056057743,0.004214717,0.00039116445,0.0001877078,0.00034919626,0.00016179017,0.04233181,0.033789616,0.0084973695,0.020063607,0.8890077],"study_design_scores_gemma":[0.000037310547,0.00018738059,0.0020207907,0.00003628083,0.000078712175,0.0003588426,0.00008628179,0.9649402,0.013400616,0.014166301,0.0046473043,0.00003995187],"about_ca_topic_score_codex":0.006216206,"about_ca_topic_score_gemma":0.014812695,"teacher_disagreement_score":0.006216206,"about_ca_system_score_codex":0.00088838587,"about_ca_system_score_gemma":0.0011245908,"threshold_uncertainty_score":0.0126369},"labels":[],"label_agreement":null},{"id":"W2798338993","doi":"10.1109/cvpr.2018.00743","title":"A Face-to-Face Neural Conversation Model","year":2018,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Conversation; Gesture; Computer science; Dialog box; Avatar; Speech recognition; Face (sociological concept); Natural (archaeology); Artificial intelligence; Human–computer interaction; Natural language processing; Communication; Psychology; Linguistics","score_opus":0.022569612808113572,"score_gpt":0.2964864715879034,"score_spread":0.27391685877978983,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2798338993","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14330278,0.0010411387,0.8246115,0.0025302446,0.00030997492,0.00031945488,0.0013551434,0.0030694765,0.023460252],"genre_scores_gemma":[0.89372337,0.00026358425,0.07893997,0.0005293504,0.00012428132,0.00046479816,0.000754643,0.000111711786,0.025088292],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995436,0.00013374246,0.000015341202,0.00017265292,0.000074394105,0.0000602278],"domain_scores_gemma":[0.9996226,0.0001717875,0.000024045956,0.000031466036,0.00012033084,0.000029738474],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007998473,0.00078168185,0.0005287187,0.0003810223,0.00060090906,0.0006622201,0.0017292423,0.0017641523,0.005723692],"category_scores_gemma":[0.0019381007,0.0003969402,0.0006895166,0.0002964092,0.0004126443,0.0013660319,0.0010234977,0.0015136387,0.0016585435],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049650297,0.0002693253,0.0018299001,0.00017147986,0.00012061273,0.0004050649,0.00049494504,0.80097693,0.013300174,0.01730707,0.008583395,0.15604465],"study_design_scores_gemma":[0.000005114075,0.000022409204,0.00011743017,0.0000034119953,0.00000890453,0.000035212455,0.00001213857,0.99632347,0.0007617092,0.0022109647,0.0004929762,0.0000061750625],"about_ca_topic_score_codex":0.012450338,"about_ca_topic_score_gemma":0.009727648,"teacher_disagreement_score":0.012450338,"about_ca_system_score_codex":0.0011695331,"about_ca_system_score_gemma":0.00089014403,"threshold_uncertainty_score":0.024755716},"labels":[],"label_agreement":null},{"id":"W2798881773","doi":"10.18653/v1/p18-1085","title":"Illustrative Language Understanding: Large-Scale Visual Grounding with Image Search","year":2018,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Google (Canada)","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing; Sentence; Word (group theory); Embedding; Image (mathematics); Semantics (computer science); Inference; Vocabulary; Speech recognition; Linguistics","score_opus":0.02645762873014901,"score_gpt":0.3338192082293743,"score_spread":0.30736157949922527,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2798881773","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34384003,0.00075950666,0.6271536,0.00091825705,0.00016182607,0.00025329395,0.0022252372,0.0103445295,0.014343753],"genre_scores_gemma":[0.86429733,0.00021773108,0.12893407,0.00020504663,0.000038047554,0.000111356756,0.001772601,0.0003426848,0.0040812143],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99975246,0.000051154613,0.000011394702,0.00010764634,0.000043650292,0.00003362742],"domain_scores_gemma":[0.99933547,0.00027015936,0.00009642572,0.00016860254,0.00005750533,0.00007178925],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00028393805,0.0007329209,0.00037854063,0.00062473904,0.00022394335,0.0012010484,0.0009481388,0.00063004706,0.008599402],"category_scores_gemma":[0.004223855,0.00029030888,0.0005027482,0.0005962154,0.00061613275,0.0051290547,0.0018888994,0.0007727974,0.0013882853],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00100512,0.00038088523,0.008939539,0.00065881433,0.00014431131,0.0007809811,0.0027674076,0.019025946,0.21531834,0.03560595,0.015939014,0.6994338],"study_design_scores_gemma":[0.000087456574,0.0008498036,0.022041546,0.00015319801,0.00015753387,0.0009137291,0.0014327477,0.6773905,0.11270777,0.1567157,0.027401054,0.00014892253],"about_ca_topic_score_codex":0.0025739833,"about_ca_topic_score_gemma":0.0034645905,"teacher_disagreement_score":0.008599402,"about_ca_system_score_codex":0.0004186799,"about_ca_system_score_gemma":0.00031157798,"threshold_uncertainty_score":0.028767884},"labels":[],"label_agreement":null},{"id":"W2799002257","doi":"10.1109/cvpr.2018.00886","title":"VirtualHome: Simulating Household Activities Via Programs","year":2018,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":350,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; Defense Advanced Research Projects Agency; Intelligence Advanced Research Projects Activity; Samsung; Nvidia","keywords":"Computer science; Task (project management); Representation (politics); Variety (cybernetics); Human–computer interaction; Interface (matter); Code (set theory); Game engine; Natural language; Artificial intelligence; Programming language","score_opus":0.02535626076205967,"score_gpt":0.2773558627617337,"score_spread":0.251999601999674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2799002257","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4472789,0.00024757153,0.52871835,0.00029914675,0.00008369324,0.0005214297,0.0042940364,0.009758864,0.00879791],"genre_scores_gemma":[0.8147359,0.00020184316,0.1768283,0.00009995038,0.000009189251,0.0005146976,0.00386981,0.0005349798,0.0032052477],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99983525,0.000065261476,0.000008251676,0.000044811182,0.000027410088,0.000019005905],"domain_scores_gemma":[0.99971646,0.00017667856,0.000019237266,0.000036594527,0.000021564083,0.000029529527],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00021781154,0.00067435886,0.00026854762,0.000353139,0.00019467481,0.0005320495,0.0011996313,0.0006568769,0.0033531266],"category_scores_gemma":[0.0011878285,0.00026081604,0.00049725256,0.00025333045,0.00059524336,0.000814601,0.0008603788,0.00056014187,0.00040253336],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005409574,0.00043618193,0.007750667,0.0004215345,0.0000813632,0.00055990205,0.0011133102,0.9030892,0.015332825,0.011371088,0.0072599105,0.05204305],"study_design_scores_gemma":[0.000048231374,0.00009694928,0.0014221834,0.000023603563,0.000014586568,0.000087942186,0.00018219402,0.98127854,0.0062896786,0.0034592885,0.007078035,0.000018712297],"about_ca_topic_score_codex":0.008309905,"about_ca_topic_score_gemma":0.011492218,"teacher_disagreement_score":0.008309905,"about_ca_system_score_codex":0.00043992803,"about_ca_system_score_gemma":0.000390291,"threshold_uncertainty_score":0.016523063},"labels":[],"label_agreement":null},{"id":"W2801004733","doi":"10.1109/wacv.2018.00181","title":"Scaling Human-Object Interaction Recognition Through Zero-Shot Learning","year":2018,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":163,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Artificial intelligence; Object (grammar); Object detection; Cognitive neuroscience of visual object recognition; Variety (cybernetics); Pattern recognition (psychology); Machine learning; Scaling; Action (physics); Histogram; Computer vision; Image (mathematics); Mathematics","score_opus":0.0710747672886309,"score_gpt":0.3634047171561183,"score_spread":0.2923299498674874,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2801004733","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20884587,0.0018383665,0.7672437,0.00042602472,0.00031537635,0.00042742793,0.0018160291,0.013558329,0.005528884],"genre_scores_gemma":[0.73484445,0.00040081458,0.24963082,0.0005094807,0.00014756103,0.0003240662,0.008638437,0.00042317546,0.00508125],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99751925,0.00039975147,0.00009331949,0.0011990074,0.0005227533,0.00026588357],"domain_scores_gemma":[0.99727005,0.001093248,0.00015469265,0.00080977817,0.0004726968,0.00019956764],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019534335,0.0014318158,0.002015017,0.0017925074,0.0007723262,0.0013349347,0.0036044007,0.0020570534,0.0021360535],"category_scores_gemma":[0.0057705157,0.0005595797,0.0011862767,0.0011705674,0.0012086746,0.0028285068,0.0034793233,0.0023991284,0.001911318],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005285001,0.0012053789,0.010967824,0.00040374004,0.00025969133,0.0003378448,0.00041291732,0.070140265,0.033743806,0.002889499,0.019715456,0.8593951],"study_design_scores_gemma":[0.000021119733,0.00014501413,0.0044900263,0.000022425756,0.000033509154,0.0002872526,0.00013545876,0.9701136,0.013651285,0.007821407,0.0032389057,0.000040062696],"about_ca_topic_score_codex":0.0077428217,"about_ca_topic_score_gemma":0.01427178,"teacher_disagreement_score":0.0077428217,"about_ca_system_score_codex":0.001128761,"about_ca_system_score_gemma":0.0010143061,"threshold_uncertainty_score":0.015395522},"labels":[],"label_agreement":null},{"id":"W2803461564","doi":"10.18653/v1/n18-2122","title":"The Emergence of Semantics in Neural Network Representations of Visual Information","year":2018,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Western Canada Research Grid; Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Computer science; Semantics (computer science); Computational semantics; Computational linguistics; Artificial intelligence; Artificial neural network; Volume (thermodynamics); Natural language processing; Cognitive science; Association (psychology); Linguistics; Programming language; Operational semantics; Psychology; Epistemology; Philosophy","score_opus":0.01069177315046837,"score_gpt":0.31955754322869806,"score_spread":0.3088657700782297,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2803461564","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6399061,0.003944005,0.32618412,0.0057444833,0.00038278152,0.000057836936,0.0009614324,0.0008494749,0.021969626],"genre_scores_gemma":[0.9782835,0.0005998647,0.019185068,0.00010603055,0.0000530799,0.000022008151,0.00024317828,0.00009459493,0.0014125748],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99982053,0.000052418847,0.000012891457,0.000054848213,0.000032798245,0.000026538446],"domain_scores_gemma":[0.9977507,0.0012620557,0.0002457751,0.0002657298,0.00035437377,0.00012146323],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005444732,0.00020441256,0.0003830708,0.0006673425,0.00040378261,0.002193403,0.0007401338,0.0007922567,0.0018304914],"category_scores_gemma":[0.010010891,0.0005304726,0.00042600924,0.0006746742,0.0013548073,0.0059772017,0.0013720035,0.001717594,0.00023729],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000654916,0.00016757064,0.008610626,0.0005698923,0.00016185052,0.00075333763,0.0024991944,0.05582275,0.084477216,0.61434525,0.009904326,0.22203319],"study_design_scores_gemma":[0.00004074273,0.00006688574,0.006434143,0.00005708805,0.000042829735,0.0003251289,0.00037310622,0.313247,0.0071909465,0.66966176,0.0025194688,0.00004083772],"about_ca_topic_score_codex":0.001973887,"about_ca_topic_score_gemma":0.002571432,"teacher_disagreement_score":0.002193403,"about_ca_system_score_codex":0.00050273526,"about_ca_system_score_gemma":0.0003102512,"threshold_uncertainty_score":0.0061236024},"labels":[],"label_agreement":null},{"id":"W2808873503","doi":"10.1109/taslp.2019.2943018","title":"BFGAN: Backward and Forward Generative Adversarial Networks for Lexically Constrained Sentence Generation","year":2019,"lang":"en","type":"article","venue":"IEEE/ACM Transactions on Audio Speech and Language Processing","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Generator (circuit theory); Discriminator; Process (computing); Sentence; Adversarial system; Generative grammar; Encoder; Joint (building)","score_opus":0.012741270686931429,"score_gpt":0.2707468446229923,"score_spread":0.2580055739360609,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2808873503","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008270391,0.0004545569,0.9857874,0.00033969234,0.00008578245,0.00009584959,0.00020604042,0.0019248141,0.0028354968],"genre_scores_gemma":[0.50929064,0.00064266264,0.47207236,0.0012313365,0.00016920104,0.0006189394,0.0016376388,0.00073858665,0.013598667],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996228,0.00017849267,0.000012185899,0.00007903926,0.000070320966,0.000037176436],"domain_scores_gemma":[0.9992059,0.00058308203,0.000040110215,0.00007530641,0.000071175586,0.000024335824],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010562325,0.0013150646,0.00064151693,0.00041070895,0.00033229747,0.00041642264,0.001289811,0.0011713036,0.0036570062],"category_scores_gemma":[0.0025192124,0.00049363816,0.00055968633,0.00031329907,0.00076220965,0.0010012158,0.0012506007,0.0019929023,0.0011833411],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016446377,0.000081595965,0.0005225955,0.00013094189,0.000078937526,0.00020739532,0.0001281731,0.79788554,0.009530429,0.02212546,0.01022101,0.15892346],"study_design_scores_gemma":[0.000010780391,0.000019527432,0.000042997057,0.000008780736,0.0000051563334,0.00003099284,0.0000047963504,0.9893801,0.0014111665,0.00804216,0.0010374632,0.000006066348],"about_ca_topic_score_codex":0.0028087078,"about_ca_topic_score_gemma":0.0053384965,"teacher_disagreement_score":0.0036570062,"about_ca_system_score_codex":0.0006138684,"about_ca_system_score_gemma":0.0007153411,"threshold_uncertainty_score":0.012233913},"labels":[],"label_agreement":null},{"id":"W2882995289","doi":"","title":"Joint Embeddings of Scene Graphs and Images","year":2017,"lang":"en","type":"preprint","venue":"Lirias (KU Leuven)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canada Research Chairs; University of Toronto","funders":"Mitacs; KU Leuven","keywords":"Computer science; Representation (politics); Joint (building); Task (project management); Artificial intelligence; Scene graph; Pattern recognition (psychology); Computer vision; Natural language processing","score_opus":0.030922476231541648,"score_gpt":0.30460035726814105,"score_spread":0.2736778810365994,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2882995289","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.062482975,0.0102684,0.73862016,0.0072992016,0.006640686,0.00028222406,0.057512667,0.03794563,0.0789481],"genre_scores_gemma":[0.39669368,0.011952704,0.28276947,0.0008921131,0.00156468,0.00023026907,0.15157081,0.009367266,0.14495915],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99914026,0.00015708026,0.000029981089,0.00036963253,0.00021092776,0.00009199595],"domain_scores_gemma":[0.9982545,0.0002695734,0.00009421314,0.0007816338,0.00047836165,0.00012170548],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007021918,0.0014805278,0.00095139234,0.0017915982,0.00041318985,0.0035187106,0.0012855148,0.0012432205,0.04475555],"category_scores_gemma":[0.004566259,0.00054267433,0.0010668397,0.0027999028,0.0006844068,0.0053226035,0.0019105059,0.0016301002,0.026880149],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051666005,0.00019882363,0.0012752301,0.00082468125,0.00018949734,0.00017219505,0.00014243094,0.030726649,0.01340493,0.05023581,0.17835367,0.7239593],"study_design_scores_gemma":[0.000078102275,0.00017430505,0.0060117156,0.00034833146,0.00018505463,0.00049750163,0.0002851544,0.38685614,0.03448434,0.15368432,0.41725126,0.00014388077],"about_ca_topic_score_codex":0.0058761863,"about_ca_topic_score_gemma":0.005500182,"teacher_disagreement_score":0.04475555,"about_ca_system_score_codex":0.00099521,"about_ca_system_score_gemma":0.0010672102,"threshold_uncertainty_score":0.14972222},"labels":[],"label_agreement":null},{"id":"W2884007780","doi":"10.71781/10727","title":"Learning visual representations with neural networks for video captioning and image generation","year":2017,"lang":"en","type":"dissertation","venue":"Open MIND","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Canadian Institute for Advanced Research","keywords":"Closed captioning; Computer science; Artificial intelligence; Artificial neural network; Image (mathematics); Natural language processing; Speech recognition; Computer vision","score_opus":0.03202932803664612,"score_gpt":0.3788374465242605,"score_spread":0.3468081184876144,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2884007780","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.062065575,0.002082645,0.9273971,0.0006571129,0.00019393476,0.00010959498,0.00020883193,0.00271703,0.004568165],"genre_scores_gemma":[0.70935917,0.0014195836,0.27778313,0.00027579695,0.00014476168,0.00015894337,0.0005489236,0.00025492432,0.010054693],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996253,0.000074471674,0.000018558834,0.00013680191,0.0000804665,0.00006429835],"domain_scores_gemma":[0.9993018,0.00035488067,0.000068226174,0.000085811786,0.00015522601,0.000033988465],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007912751,0.0009056178,0.00065768935,0.0006874668,0.00034647583,0.001273332,0.0014531673,0.0014152445,0.002935743],"category_scores_gemma":[0.0031281684,0.00057513226,0.00087976,0.00076084386,0.00055707106,0.0021515826,0.0008735482,0.0018793778,0.0006612785],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036213844,0.00018743328,0.0010276781,0.0002792516,0.00015132732,0.0001980298,0.00019502119,0.48411056,0.03974529,0.012250139,0.0033947537,0.45809832],"study_design_scores_gemma":[0.00000502883,0.000026846259,0.00019886061,0.000008590823,0.000011555806,0.00002017928,0.000016535478,0.9912702,0.004381468,0.0034002713,0.000653074,0.0000073642977],"about_ca_topic_score_codex":0.00891651,"about_ca_topic_score_gemma":0.009234591,"teacher_disagreement_score":0.00891651,"about_ca_system_score_codex":0.0012980199,"about_ca_system_score_gemma":0.00066719146,"threshold_uncertainty_score":0.017729223},"labels":[],"label_agreement":null},{"id":"W2886247548","doi":"10.1007/978-3-030-01228-1_48","title":"Visual Reasoning with Multi-hop Feature Modulation","year":2018,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Institute for Advanced Research; Université de Montréal","funders":"","keywords":"Computer science; Artificial intelligence; Computation; Feature (linguistics); Question answering; Natural language processing; Hierarchy; Algorithm","score_opus":0.013293937599181143,"score_gpt":0.279478282932439,"score_spread":0.26618434533325785,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2886247548","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011846543,0.00048879505,0.975546,0.00023541089,0.000116692754,0.00005283232,0.00012867028,0.0012273022,0.010357768],"genre_scores_gemma":[0.5856269,0.00073685485,0.4010762,0.0002552852,0.00015081039,0.00011532671,0.00045277146,0.00020408691,0.011381769],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99968815,0.000051432737,0.000020590847,0.000099582285,0.000103947925,0.00003638043],"domain_scores_gemma":[0.9995154,0.00024526886,0.00003289645,0.000119536104,0.00006255868,0.000024375166],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003433247,0.0005987381,0.00046498628,0.0005512712,0.00030631493,0.0014048483,0.0014078632,0.000863949,0.008664641],"category_scores_gemma":[0.00248175,0.0002897636,0.0006135809,0.0005442182,0.0005513684,0.0026088334,0.0013646459,0.0010212905,0.001418526],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043274692,0.00017022344,0.00033160765,0.00033211117,0.00007162264,0.00026594015,0.00018938578,0.050273586,0.047182817,0.114936456,0.0098651005,0.77594846],"study_design_scores_gemma":[0.000044945038,0.00007606145,0.00036153037,0.000043557244,0.00004129363,0.00022571126,0.000055611617,0.6872046,0.022862252,0.28141305,0.007645264,0.000026196965],"about_ca_topic_score_codex":0.0012018203,"about_ca_topic_score_gemma":0.00086454686,"teacher_disagreement_score":0.008664641,"about_ca_system_score_codex":0.00044313268,"about_ca_system_score_gemma":0.00027010523,"threshold_uncertainty_score":0.028986096},"labels":[],"label_agreement":null},{"id":"W2891997820","doi":"10.48550/arxiv.1810.11735","title":"Middle-Out Decoding","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Closed captioning; Decoding methods; Computer science; Sequence (biology); Task (project management); Word (group theory); Process (computing); Speech recognition; Dual (grammatical number); Artificial intelligence; Algorithm; Linguistics; Image (mathematics); Engineering; Programming language","score_opus":0.12850826277725302,"score_gpt":0.21877874903884442,"score_spread":0.0902704862615914,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2891997820","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024572924,0.0004408921,0.9560795,0.00041666898,0.00027335098,0.00012128015,0.00026725573,0.0044805775,0.01334758],"genre_scores_gemma":[0.607116,0.00044782463,0.3619643,0.00092521776,0.00022448608,0.00026814043,0.0011257601,0.0017050214,0.026223205],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99922,0.00019989227,0.000056466688,0.00025337015,0.00018477056,0.00008553678],"domain_scores_gemma":[0.9979588,0.00093331287,0.000091327944,0.0005101468,0.00040160547,0.00010477843],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011493898,0.0012020819,0.00086386176,0.00067071157,0.00064559263,0.0016380878,0.0015231647,0.0015372772,0.010406089],"category_scores_gemma":[0.006029211,0.00040275021,0.00063542434,0.00046416017,0.0011233921,0.002651687,0.0021532164,0.001725465,0.0032794632],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011725276,0.00025861987,0.001678704,0.00037589303,0.00009368887,0.00067958276,0.00088532094,0.150758,0.09006632,0.109760016,0.01636768,0.6279037],"study_design_scores_gemma":[0.00003167142,0.00012825019,0.00021792302,0.00003566705,0.00002962915,0.00026541736,0.000060509963,0.8879152,0.051645227,0.049446095,0.010189287,0.00003519411],"about_ca_topic_score_codex":0.0019768595,"about_ca_topic_score_gemma":0.0028285447,"teacher_disagreement_score":0.010406089,"about_ca_system_score_codex":0.0006948688,"about_ca_system_score_gemma":0.001204225,"threshold_uncertainty_score":0.034811795},"labels":[],"label_agreement":null},{"id":"W2894280539","doi":"10.1609/aaai.v33i01.33019062","title":"Multilevel Language and Vision Integration for Text-to-Clip Retrieval","year":2019,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":332,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Intelligence Advanced Research Projects Activity","keywords":"Computer science; Task (project management); CLIPS; Matching (statistics); Metric (unit); Natural language processing; Artificial intelligence; Sentence; Similarity (geometry); Word (group theory); Recurrent neural network; Artificial neural network; Information retrieval; Image (mathematics)","score_opus":0.04201553110016543,"score_gpt":0.34819292535200913,"score_spread":0.3061773942518437,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2894280539","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.057284188,0.0027372052,0.90635544,0.00084893784,0.00036686813,0.0004311829,0.0016413633,0.021713419,0.00862141],"genre_scores_gemma":[0.5687347,0.00094293576,0.40880474,0.0009682454,0.00053055666,0.0004781554,0.00626005,0.00079696666,0.012483688],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99926656,0.00010955751,0.000043964552,0.00024500606,0.00020184195,0.00013316506],"domain_scores_gemma":[0.99906987,0.0003108155,0.00007922149,0.00022136346,0.00022903286,0.0000898025],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009210044,0.0013116053,0.0013568496,0.0016681452,0.0004621608,0.0016207433,0.0024309382,0.0018646588,0.007957785],"category_scores_gemma":[0.0034836694,0.00035095096,0.0012286222,0.0016115706,0.00052714517,0.0032638563,0.0016944661,0.0021306674,0.0040592374],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00068142265,0.0008924959,0.001006175,0.0005546804,0.00019224863,0.00035529045,0.00016375554,0.12131173,0.07133438,0.008834662,0.03320715,0.7614659],"study_design_scores_gemma":[0.00003350159,0.00023881381,0.00035300956,0.000013935604,0.000045357745,0.00010264458,0.000028980685,0.9749228,0.015436311,0.0053259055,0.003473097,0.000025674117],"about_ca_topic_score_codex":0.009208652,"about_ca_topic_score_gemma":0.014184683,"teacher_disagreement_score":0.009208652,"about_ca_system_score_codex":0.0013066273,"about_ca_system_score_gemma":0.0012550958,"threshold_uncertainty_score":0.02662146},"labels":[],"label_agreement":null},{"id":"W2894405212","doi":"10.1167/18.10.136","title":"Totally-Looks-Like: A Dataset and Benchmark of Semantic Image Similarity","year":2018,"lang":"en","type":"article","venue":"Journal of Vision","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Computer science; Similarity (geometry); Artificial intelligence; Representation (politics); Perception; Sketch; Image (mathematics); Semantics (computer science); Pattern recognition (psychology); Information retrieval; Psychology","score_opus":0.009170502743326061,"score_gpt":0.3133320988807346,"score_spread":0.30416159613740856,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2894405212","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43855625,0.023074433,0.021373888,0.0025109602,0.0031648956,0.002811182,0.4547777,0.017296363,0.03643443],"genre_scores_gemma":[0.19268456,0.0017347948,0.039373226,0.0009281574,0.0003705368,0.0007065869,0.7581596,0.00060964224,0.0054329],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99674666,0.0005736684,0.0003668128,0.001033474,0.000994378,0.00028492452],"domain_scores_gemma":[0.99670297,0.0007214677,0.00037382077,0.0009963278,0.0006212277,0.00058428326],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015023127,0.0040291087,0.001811948,0.005861408,0.0017170649,0.002516147,0.004517716,0.0045914375,0.007400877],"category_scores_gemma":[0.007124228,0.00042170912,0.0039240927,0.004599486,0.0015108688,0.0030815967,0.003544226,0.0030148705,0.005832474],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004985911,0.004505516,0.037182182,0.009753759,0.0031989624,0.0028880204,0.00094362965,0.025331084,0.020607896,0.005457142,0.61841005,0.26673588],"study_design_scores_gemma":[0.0021287648,0.006893943,0.18593334,0.002294894,0.0013625601,0.01974582,0.0049599647,0.26930046,0.034289166,0.021132847,0.45116013,0.000798149],"about_ca_topic_score_codex":0.0150488615,"about_ca_topic_score_gemma":0.028078679,"teacher_disagreement_score":0.0150488615,"about_ca_system_score_codex":0.0022483275,"about_ca_system_score_gemma":0.0013111233,"threshold_uncertainty_score":0.029922545},"labels":[],"label_agreement":null},{"id":"W2900379270","doi":"10.18653/v1/w18-6240","title":"EmojiGAN: learning emojis distributions with a generative model","year":2018,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Generative grammar; Adversarial system; Emoji; Artificial intelligence; Popularity; Machine learning; Image (mathematics); Set (abstract data type); Task (project management); Generative model; Natural language processing; Social media","score_opus":0.012900485235033779,"score_gpt":0.2699766042742528,"score_spread":0.257076119039219,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2900379270","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025982706,0.0004587112,0.96575135,0.00034609987,0.00011931168,0.000081332,0.00028685873,0.0028798727,0.004093733],"genre_scores_gemma":[0.61626494,0.0008009484,0.35669178,0.0013555324,0.00025418645,0.00038792883,0.0024636479,0.0009208781,0.020860141],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99975413,0.00009464523,0.000006684954,0.00007183954,0.000042176456,0.000030546802],"domain_scores_gemma":[0.99941826,0.00037244015,0.0000345798,0.00008413407,0.00005845537,0.00003209753],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008191999,0.0011033756,0.0007055518,0.0006059775,0.0003371395,0.0006402059,0.0010689725,0.0011453306,0.0037823622],"category_scores_gemma":[0.0022615446,0.00058549753,0.00091115583,0.00047740858,0.00067148963,0.0010127858,0.0011732761,0.0019202078,0.0020371112],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000310257,0.00020637496,0.0025481756,0.0001508527,0.00021711503,0.00030208591,0.00018424784,0.5890421,0.009787877,0.028173942,0.019182425,0.3498945],"study_design_scores_gemma":[0.000010580063,0.000023298735,0.00023225452,0.000010515754,0.00001061971,0.00005899268,0.000007859309,0.9904208,0.0012784796,0.006752778,0.001185527,0.00000820514],"about_ca_topic_score_codex":0.002514381,"about_ca_topic_score_gemma":0.0057776654,"teacher_disagreement_score":0.0037823622,"about_ca_system_score_codex":0.0005299043,"about_ca_system_score_gemma":0.00041422213,"threshold_uncertainty_score":0.012653232},"labels":[],"label_agreement":null},{"id":"W2900626451","doi":"10.1109/cvprw.2018.00260","title":"Image Caption Generation with Hierarchical Contextual Visual Spatial Attention","year":2018,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Leverage (statistics); Artificial intelligence; Grid; Word (group theory); Context (archaeology); Spatial contextual awareness; Layer (electronics); Context model; Image (mathematics); Object (grammar); Pattern recognition (psychology); Computer vision; Natural language processing; Linguistics; Geography","score_opus":0.013829384278467263,"score_gpt":0.28997068423961525,"score_spread":0.276141299961148,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2900626451","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031340178,0.0026182868,0.91991687,0.00079284795,0.00076718855,0.00027222902,0.0011841848,0.03132522,0.011783078],"genre_scores_gemma":[0.49774265,0.0013868838,0.47735435,0.001322031,0.00037559308,0.00040397915,0.004217874,0.0012532456,0.0159434],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997279,0.000036874015,0.000012608073,0.00011495169,0.000057886955,0.00004964388],"domain_scores_gemma":[0.9996325,0.00010391726,0.000032865086,0.00008877654,0.00010646988,0.000035478],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00038938553,0.0016871619,0.00073314446,0.0006532456,0.00034584154,0.00091396115,0.001997471,0.0014463905,0.006835502],"category_scores_gemma":[0.0017868791,0.00045727167,0.00092928624,0.00070480886,0.00055530417,0.001950887,0.0014350571,0.0017490144,0.00232127],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038673996,0.00025559854,0.00066584884,0.00055298075,0.00016978732,0.0005377731,0.00029533438,0.12879862,0.08012357,0.007281221,0.038686946,0.7422456],"study_design_scores_gemma":[0.000038081518,0.00010827888,0.00038176953,0.00003140789,0.00006835462,0.00017958414,0.00004239736,0.9455212,0.034332916,0.009402992,0.009863,0.00003002208],"about_ca_topic_score_codex":0.0054995064,"about_ca_topic_score_gemma":0.0073298807,"teacher_disagreement_score":0.006835502,"about_ca_system_score_codex":0.00093475764,"about_ca_system_score_gemma":0.00076241704,"threshold_uncertainty_score":0.022866964},"labels":[],"label_agreement":null},{"id":"W2901786679","doi":"","title":"Keep Drawing It: Iterative language-based image generation and editing.","year":2018,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; University of Guelph","funders":"","keywords":"Computer science; Focus (optics); Image (mathematics); Extension (predicate logic); Context (archaeology); Image editing; Simple (philosophy); Artificial intelligence; Computer vision; Human–computer interaction; Programming language","score_opus":0.037735808935367036,"score_gpt":0.21462178105384494,"score_spread":0.1768859721184779,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2901786679","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004861048,0.00012178793,0.9854886,0.00012995626,0.000057321864,0.00008017566,0.00008090435,0.006612617,0.0025675122],"genre_scores_gemma":[0.21226507,0.00020039541,0.774566,0.00024000778,0.0000625964,0.00024317644,0.00050766155,0.0020570068,0.009858034],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989581,0.00033952438,0.00004087145,0.00028584164,0.00029835093,0.00007731578],"domain_scores_gemma":[0.996915,0.0017782795,0.0001759036,0.0006341603,0.00032931197,0.00016738671],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015417638,0.0010744198,0.0005953534,0.0005936832,0.00047253587,0.0013083097,0.0036296973,0.0015633397,0.008117258],"category_scores_gemma":[0.0072761704,0.000596368,0.0010829138,0.00041120453,0.0010960575,0.0019928748,0.002360332,0.0014649398,0.0032727614],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055015855,0.0004109054,0.0013644159,0.0004355596,0.00017789371,0.0011550182,0.0021324072,0.17867194,0.09008713,0.056450628,0.03229034,0.6362736],"study_design_scores_gemma":[0.00004307437,0.00007327288,0.00017685721,0.0000201085,0.00002643732,0.00026744528,0.000069319205,0.94166934,0.02348643,0.024561021,0.009567342,0.000039423736],"about_ca_topic_score_codex":0.0019464265,"about_ca_topic_score_gemma":0.0031356674,"teacher_disagreement_score":0.008117258,"about_ca_system_score_codex":0.00051993784,"about_ca_system_score_gemma":0.00061074423,"threshold_uncertainty_score":0.027154922},"labels":[],"label_agreement":null},{"id":"W2903711666","doi":"10.1109/tpami.2018.2886192","title":"Deep Neural Network Compression by In-Parallel Pruning-Quantization","year":2018,"lang":"en","type":"article","venue":"IEEE Transactions on Pattern Analysis and Machine Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":138,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Pruning; Quantization (signal processing); Convolutional neural network; Deep learning; Artificial intelligence; Artificial neural network; Pattern recognition (psychology); Computer vision","score_opus":0.01616013985227046,"score_gpt":0.2839327377264118,"score_spread":0.26777259787414137,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2903711666","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1325091,0.0015959371,0.8544006,0.0006536749,0.0002559248,0.0002210232,0.00043110378,0.005631589,0.0043010265],"genre_scores_gemma":[0.6540109,0.0006636314,0.33806747,0.000520229,0.0001348954,0.00023889913,0.0013516782,0.0004089148,0.0046033734],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994886,0.00006879965,0.000047837126,0.000098383505,0.00023692987,0.000059477512],"domain_scores_gemma":[0.9986313,0.00037688125,0.00011788349,0.00048358,0.00034409642,0.000046315225],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00083031284,0.0012400356,0.0008881303,0.00087182206,0.00049405027,0.0009779582,0.0018218888,0.00079734134,0.00232412],"category_scores_gemma":[0.004513705,0.00033753007,0.0005309425,0.0010109494,0.00081074145,0.0019123554,0.0012264778,0.0016485143,0.0006459576],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052252307,0.00035264943,0.0036165304,0.00022137692,0.0001257679,0.00033551187,0.00024959032,0.2746093,0.03844994,0.015317071,0.012716777,0.6534829],"study_design_scores_gemma":[0.000062748746,0.0001584146,0.00069520756,0.000028121516,0.000041927196,0.0002043812,0.000055526347,0.947582,0.034737736,0.012475856,0.0039410405,0.000017075865],"about_ca_topic_score_codex":0.004675454,"about_ca_topic_score_gemma":0.008577348,"teacher_disagreement_score":0.004675454,"about_ca_system_score_codex":0.00089546293,"about_ca_system_score_gemma":0.0012210552,"threshold_uncertainty_score":0.009296477},"labels":[],"label_agreement":null},{"id":"W2905288264","doi":"10.1609/aaai.v33i01.33018885","title":"Connecting Language to Images: A Progressive Attention-Guided Network for Simultaneous Image Captioning and Language Grounding","year":2019,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Science and Technology Planning Project of Guangdong Province; National Key Research and Development Program of China; China Scholarship Council; National Natural Science Foundation of China","keywords":"Closed captioning; Bounding overwatch; Computer science; Benchmark (surveying); Artificial intelligence; Image (mathematics); Process (computing); Machine learning; Natural language processing; Pattern recognition (psychology)","score_opus":0.030288974158012197,"score_gpt":0.3297025353385947,"score_spread":0.29941356118058254,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2905288264","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0812729,0.0014248148,0.898527,0.0012199011,0.00029218214,0.00027863763,0.0007252059,0.008442795,0.007816549],"genre_scores_gemma":[0.74299693,0.000543451,0.24355431,0.0009871739,0.00019443982,0.00028518698,0.0018756883,0.00039284592,0.009169958],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995703,0.0000939907,0.000014730167,0.00019912788,0.000058557507,0.00006325251],"domain_scores_gemma":[0.999183,0.0003745973,0.00007771824,0.00013874634,0.00015466836,0.00007128549],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007669594,0.0016826398,0.00085379544,0.00097554096,0.00060608407,0.0009037572,0.0031211046,0.0022928263,0.0031614678],"category_scores_gemma":[0.0031564615,0.0006532161,0.0010016155,0.0007245616,0.0012024649,0.0027478652,0.0018372649,0.0020803697,0.0009895161],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005802085,0.00042144395,0.0020557255,0.00033831163,0.00015122515,0.00058855687,0.00044857492,0.44230813,0.031567268,0.015483727,0.016953098,0.48910376],"study_design_scores_gemma":[0.00001840815,0.000055257555,0.00015552544,0.000013406665,0.000022827657,0.000055069533,0.000022013943,0.9843993,0.005132113,0.008855812,0.0012576502,0.000012587552],"about_ca_topic_score_codex":0.0091061825,"about_ca_topic_score_gemma":0.010714678,"teacher_disagreement_score":0.0091061825,"about_ca_system_score_codex":0.0013762101,"about_ca_system_score_gemma":0.001092637,"threshold_uncertainty_score":0.018106341},"labels":[],"label_agreement":null},{"id":"W2913694129","doi":"10.1145/3279952","title":"Deep Learning–Based Multimedia Analytics","year":2019,"lang":"en","type":"article","venue":"ACM Transactions on Multimedia Computing Communications and Applications","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"National Laboratory of Pattern Recognition; National Natural Science Foundation of China","keywords":"Computer science; Deep learning; Analytics; Closed captioning; Multimedia; Milestone; Domain (mathematical analysis); Visual analytics; Learning analytics; Data science; Artificial intelligence; Visualization; Image (mathematics)","score_opus":0.017483534786770713,"score_gpt":0.2872168292985619,"score_spread":0.2697332945117912,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2913694129","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02122949,0.024597578,0.92329645,0.002917834,0.00046256173,0.00014680292,0.0027477385,0.006433748,0.018167723],"genre_scores_gemma":[0.50879145,0.04678361,0.41374308,0.0015430246,0.0010714233,0.0002763747,0.00893914,0.0006525815,0.01819938],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995115,0.000084133906,0.000031936062,0.00010167017,0.00021334193,0.00005741873],"domain_scores_gemma":[0.99936265,0.00028757864,0.000047407753,0.00007417047,0.00019610784,0.00003212343],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00069363124,0.0011288228,0.0006079635,0.0017892409,0.00022542794,0.0018605995,0.0012592688,0.0007585829,0.0039169644],"category_scores_gemma":[0.0028427974,0.00029954637,0.0005802626,0.001578994,0.0005013723,0.0029544975,0.0013210714,0.0017408876,0.0019343136],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024670723,0.00014333535,0.0018888857,0.0007160963,0.000139648,0.0001496799,0.0000996989,0.08825563,0.013516547,0.03210711,0.029744916,0.8329918],"study_design_scores_gemma":[0.000018841734,0.000084247564,0.0011194573,0.00019858568,0.000055843146,0.00012391292,0.000073088275,0.8821883,0.018736212,0.06299064,0.03437229,0.00003859909],"about_ca_topic_score_codex":0.00275211,"about_ca_topic_score_gemma":0.0036267892,"teacher_disagreement_score":0.0039169644,"about_ca_system_score_codex":0.0008466492,"about_ca_system_score_gemma":0.0006856142,"threshold_uncertainty_score":0.013103604},"labels":[],"label_agreement":null},{"id":"W2933890497","doi":"10.1109/iccv.2019.01040","title":"Tell, Draw, and Repeat: Generating and Modifying Images Based on Continual Linguistic Instruction","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Institute for Advanced Research; Université de Montréal; University of Guelph; Vector Institute","funders":"Agence Nationale de la Recherche","keywords":"Computer science; Extension (predicate logic); Image (mathematics); Code (set theory); Simple (philosophy); Code generation; Artificial intelligence; Natural language processing; Computer vision; Programming language; Key (lock)","score_opus":0.013444786845761662,"score_gpt":0.27378596282388734,"score_spread":0.26034117597812567,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2933890497","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11642601,0.00035913294,0.8600701,0.0006317296,0.00015094121,0.0001378678,0.00022582829,0.008873508,0.013124742],"genre_scores_gemma":[0.73994374,0.000262433,0.242195,0.00021644963,0.000041714495,0.00012952178,0.000369524,0.0010856978,0.015755948],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99977237,0.00005947568,0.000008370103,0.00008304527,0.000052444175,0.000024337527],"domain_scores_gemma":[0.9990017,0.00057048805,0.000075562646,0.0001971674,0.00008534147,0.000069819806],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00053121796,0.00059129146,0.0003680068,0.00032047636,0.00032220266,0.0008986862,0.0015797406,0.00089485047,0.007010843],"category_scores_gemma":[0.0034749168,0.0003884731,0.00066389755,0.00019413706,0.0009835252,0.001635898,0.0009616677,0.0008714981,0.0013437382],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00082548027,0.00029098877,0.0030604252,0.0003757402,0.00014317391,0.0010935492,0.0017965593,0.24271178,0.08247031,0.080969684,0.013653142,0.5726091],"study_design_scores_gemma":[0.00004098985,0.00009293323,0.00053920853,0.000019288385,0.00003650399,0.00022082873,0.00007512846,0.9359552,0.026159044,0.03091458,0.00591473,0.00003147795],"about_ca_topic_score_codex":0.0027378325,"about_ca_topic_score_gemma":0.0032784285,"teacher_disagreement_score":0.007010843,"about_ca_system_score_codex":0.0005680077,"about_ca_system_score_gemma":0.0003630946,"threshold_uncertainty_score":0.023453593},"labels":[],"label_agreement":null},{"id":"W2944195906","doi":"10.1109/ipas.2018.8708893","title":"A Hierarchical Quasi-Recurrent approach to Video Captioning","year":2018,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Closed captioning; Computer science; Encoding (memory); Encoder; Recurrent neural network; Exploit; Representation (politics); Decoding methods; Artificial intelligence; Layer (electronics); Detector; Artificial neural network; Computer vision; Image (mathematics); Algorithm; Telecommunications","score_opus":0.022680602477925558,"score_gpt":0.2990787251192561,"score_spread":0.2763981226413305,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2944195906","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009469948,0.0003798689,0.98673964,0.00014127004,0.00005281439,0.00005325568,0.0002360147,0.0018131598,0.0011141269],"genre_scores_gemma":[0.37935102,0.0006174093,0.60889703,0.00026878054,0.00017255968,0.00018928421,0.001972748,0.00043221234,0.008098922],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99939585,0.00020458331,0.000032299227,0.00018453678,0.00011761123,0.00006516011],"domain_scores_gemma":[0.99921477,0.00028295355,0.0000845519,0.00016549676,0.00021346827,0.000038713595],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00082218694,0.00082298566,0.00062013086,0.000802379,0.00030805188,0.0008300819,0.0018195817,0.0010896235,0.0034116893],"category_scores_gemma":[0.002707886,0.00045455675,0.0008446496,0.0007502296,0.0005117418,0.0015930663,0.000958674,0.0013282468,0.0013670862],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035786934,0.00019571434,0.0006391982,0.00035145294,0.00013689828,0.0003627753,0.0003892033,0.26636553,0.0824413,0.022399705,0.011706466,0.6146539],"study_design_scores_gemma":[0.000006431459,0.00005236077,0.00015049969,0.000008509426,0.000015813372,0.00005635208,0.000016041407,0.9855527,0.007803214,0.004786811,0.0015396791,0.000011479627],"about_ca_topic_score_codex":0.0042558922,"about_ca_topic_score_gemma":0.006967444,"teacher_disagreement_score":0.0042558922,"about_ca_system_score_codex":0.000684308,"about_ca_system_score_gemma":0.0005336974,"threshold_uncertainty_score":0.011413217},"labels":[],"label_agreement":null},{"id":"W2945922950","doi":"10.1007/978-3-030-20876-9_42","title":"Dynamic Gated Graph Neural Networks for Scene Graph Generation","year":2019,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Graph; Artificial neural network; Artificial intelligence; Theoretical computer science","score_opus":0.014895537652651124,"score_gpt":0.26298440261584144,"score_spread":0.24808886496319033,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2945922950","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009508894,0.00040124333,0.98303795,0.00017062404,0.00008423376,0.00007966669,0.00042051682,0.0030654348,0.0032315117],"genre_scores_gemma":[0.32691392,0.0007814813,0.6521473,0.0003606818,0.00011512899,0.00023777997,0.0029930484,0.0013001898,0.015150395],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99986196,0.00002422554,0.0000046885443,0.00005363483,0.000033932847,0.000021555663],"domain_scores_gemma":[0.9997639,0.00010240084,0.000015101345,0.00004998549,0.000049655613,0.000018875295],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002305751,0.00082649983,0.0007069858,0.0007822802,0.00030855215,0.0006431794,0.00185944,0.0010504439,0.008461827],"category_scores_gemma":[0.0008752996,0.0005630192,0.00065054704,0.0011306803,0.0004026347,0.0010040699,0.0009587142,0.0015102912,0.0018431874],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000120959354,0.000104386905,0.00020444338,0.0000894249,0.00004644283,0.00007232379,0.000033197743,0.5387787,0.013487353,0.016551195,0.011993795,0.41851774],"study_design_scores_gemma":[0.0000036477097,0.0000070290257,0.000038309146,0.0000033370345,0.0000036187466,0.0000076083547,0.0000029823373,0.9934962,0.0012168011,0.0045308964,0.0006868669,0.0000027187452],"about_ca_topic_score_codex":0.009890123,"about_ca_topic_score_gemma":0.018337099,"teacher_disagreement_score":0.009890123,"about_ca_system_score_codex":0.00088847225,"about_ca_system_score_gemma":0.0006622772,"threshold_uncertainty_score":0.028307676},"labels":[],"label_agreement":null},{"id":"W2947973444","doi":"10.18653/v1/d19-6401","title":"Structure Learning for Neural Module Networks","year":2019,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Artificial neural network; Task (project management); Class (philosophy); Simple (philosophy); Sensitivity (control systems); Deep learning","score_opus":0.006695838380215307,"score_gpt":0.24943788624816196,"score_spread":0.24274204786794665,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2947973444","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014164749,0.0004168259,0.9829595,0.0002828765,0.000021838114,0.00005556851,0.00007328317,0.0004466365,0.0015787338],"genre_scores_gemma":[0.6958609,0.00079536944,0.29619882,0.000337923,0.0001307689,0.0005754146,0.0005496576,0.00020095422,0.005350218],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99947447,0.00021784438,0.000022732118,0.00011486995,0.0001192668,0.000050794366],"domain_scores_gemma":[0.99701995,0.0021505097,0.0001912808,0.00022837773,0.0003505579,0.00005936298],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015747299,0.0007024729,0.00082977617,0.0007609037,0.00037994212,0.00079058966,0.0013930293,0.0013011359,0.0026482618],"category_scores_gemma":[0.00916582,0.0006119299,0.00071129034,0.00064740627,0.0009412025,0.002055071,0.0013286688,0.0019694075,0.0005016761],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000044216446,0.000034452358,0.00054439646,0.00010557534,0.000061463084,0.000028744405,0.0000523741,0.8761198,0.001704249,0.045688204,0.0011172463,0.07449927],"study_design_scores_gemma":[0.0000042525617,0.000012404616,0.000059683036,0.000006016797,0.0000033970405,0.000005615288,0.000002874849,0.9607938,0.00041340757,0.0384074,0.0002876179,0.0000034967377],"about_ca_topic_score_codex":0.0023162067,"about_ca_topic_score_gemma":0.0029900942,"teacher_disagreement_score":0.0026482618,"about_ca_system_score_codex":0.0014585805,"about_ca_system_score_gemma":0.00074874546,"threshold_uncertainty_score":0.010582805},"labels":[],"label_agreement":null},{"id":"W2949833607","doi":"10.48550/arxiv.1307.0414","title":"Challenges in Representation Learning: A report on three machine learning contests","year":2013,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":103,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Representation (politics); Computer science; Artificial intelligence; Learning to learn; Feature learning; Active learning (machine learning); Data science; Machine learning; Mathematics education; Psychology; Political science","score_opus":0.1648509459164146,"score_gpt":0.25149987939107327,"score_spread":0.08664893347465866,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2949833607","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09354947,0.15136382,0.16234341,0.3346093,0.067626506,0.0022454418,0.038773675,0.009459716,0.14002877],"genre_scores_gemma":[0.35387444,0.04825884,0.1576356,0.034225803,0.03868993,0.0035506093,0.18868367,0.007864237,0.16721684],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.9747791,0.0061025945,0.001261809,0.0026346254,0.0110973725,0.00412448],"domain_scores_gemma":[0.9414866,0.015965922,0.0012654932,0.0048578423,0.021295246,0.015128929],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.043119516,0.0033530535,0.0038178167,0.0049269036,0.005567324,0.013592337,0.004958752,0.005165655,0.012902484],"category_scores_gemma":[0.045630883,0.00081804424,0.002760589,0.0070270607,0.0036970465,0.010821093,0.0167286,0.0101622725,0.00813999],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024588782,0.0003028709,0.001258136,0.0005852444,0.000082351566,0.000117195304,0.00038521708,0.0018832377,0.0008024549,0.013325868,0.87338036,0.107631184],"study_design_scores_gemma":[0.00018073736,0.00027253077,0.0066310833,0.0005565476,0.00007822686,0.0004337794,0.0018072833,0.011720587,0.003921942,0.048260972,0.92593956,0.0001967062],"about_ca_topic_score_codex":0.016944515,"about_ca_topic_score_gemma":0.02428427,"teacher_disagreement_score":0.043119516,"about_ca_system_score_codex":0.006799866,"about_ca_system_score_gemma":0.009620399,"threshold_uncertainty_score":0.22804052},"labels":[],"label_agreement":null},{"id":"W2949906242","doi":"10.48550/arxiv.1612.01033","title":"Areas of Attention for Image Captioning","year":2016,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Closed captioning; Computer science; Pairwise comparison; Transformer; Artificial intelligence; Language model; Image (mathematics); Convolutional neural network; Natural language processing; Pattern recognition (psychology); Machine learning","score_opus":0.05277417207380759,"score_gpt":0.21515213246116632,"score_spread":0.16237796038735874,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2949906242","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03720162,0.0018992791,0.9354842,0.00071255985,0.0003038841,0.00023991938,0.0010538336,0.014016658,0.0090881325],"genre_scores_gemma":[0.6507046,0.0010736511,0.33048144,0.0006175373,0.00029314798,0.00031551337,0.0030221087,0.0011959234,0.012296065],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992136,0.00023514335,0.000034428376,0.0002926116,0.0001276083,0.000096635624],"domain_scores_gemma":[0.9988127,0.00043961225,0.00012290255,0.00032527867,0.00022613886,0.00007343013],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010197874,0.0013825519,0.0006283021,0.0013434797,0.00047833417,0.0013958532,0.0018501133,0.0014172718,0.0056376886],"category_scores_gemma":[0.0035562448,0.0005125077,0.0012971468,0.0009937893,0.0009821623,0.003005633,0.001903898,0.0020768247,0.0020407098],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007128426,0.00024501045,0.0027630448,0.0005212643,0.00019342962,0.00031116404,0.0005170765,0.16632032,0.049333353,0.028723683,0.030366449,0.71999234],"study_design_scores_gemma":[0.000021345579,0.000095992975,0.00084577367,0.000029128185,0.00004485956,0.00014703049,0.000049124545,0.942407,0.023579538,0.02130451,0.011451577,0.000024176732],"about_ca_topic_score_codex":0.004565594,"about_ca_topic_score_gemma":0.006174543,"teacher_disagreement_score":0.0056376886,"about_ca_system_score_codex":0.0012444177,"about_ca_system_score_gemma":0.00078340445,"threshold_uncertainty_score":0.018859923},"labels":[],"label_agreement":null},{"id":"W2950120176","doi":"10.48550/arxiv.1705.08844","title":"How a General-Purpose Commonsense Ontology can Improve Performance of Learning-Based Image Retrieval","year":2017,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Commonsense knowledge; Computer science; Exploit; Ontology; Commonsense reasoning; Artificial intelligence; Information retrieval; Natural language processing; Question answering; Testbed; Benchmark (surveying); Sentence; Knowledge retrieval; Knowledge representation and reasoning; Knowledge extraction; World Wide Web","score_opus":0.04332595075993373,"score_gpt":0.2097428559751148,"score_spread":0.16641690521518107,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950120176","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3676319,0.0081365295,0.5497443,0.0039489227,0.00088556163,0.0007810442,0.0045580566,0.034916274,0.029397389],"genre_scores_gemma":[0.67618304,0.0013659217,0.30630058,0.0009289739,0.000127547,0.00013430456,0.010261814,0.00050699216,0.0041908324],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981437,0.00040452,0.00019509552,0.00049421744,0.0005552987,0.00020718081],"domain_scores_gemma":[0.9977089,0.0008082555,0.00010613041,0.0008503735,0.00043125436,0.00009506648],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031900743,0.0013911308,0.0010885332,0.0025356347,0.0010509439,0.0018913148,0.001825638,0.0018944254,0.0040887473],"category_scores_gemma":[0.009339235,0.00031843307,0.0011791202,0.0021925452,0.00088176026,0.0077662743,0.0021802885,0.001765651,0.0022001623],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006019224,0.0006616109,0.0056432355,0.0007407088,0.00025874624,0.0002104236,0.00029799007,0.039515067,0.03140802,0.010706528,0.035450663,0.8745051],"study_design_scores_gemma":[0.0001620587,0.0005561428,0.004634811,0.0001346947,0.0003344613,0.000596834,0.0007425315,0.8373792,0.063586995,0.04759942,0.044156533,0.00011626145],"about_ca_topic_score_codex":0.016045988,"about_ca_topic_score_gemma":0.019765688,"teacher_disagreement_score":0.016045988,"about_ca_system_score_codex":0.001699358,"about_ca_system_score_gemma":0.0018202185,"threshold_uncertainty_score":0.031905174},"labels":[],"label_agreement":null},{"id":"W2950788030","doi":"10.48550/arxiv.1701.04693","title":"Incremental Learning for Robot Perception through HRI","year":2017,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Artificial intelligence; Convolutional neural network; Robot; Computer science; Perception; Robot learning; Task (project management); Deep learning; Robotics; Object (grammar); Computer vision; Cognitive neuroscience of visual object recognition; Object detection; Human–robot interaction; Human–computer interaction; Mobile robot; Pattern recognition (psychology); Engineering","score_opus":0.11343942762562877,"score_gpt":0.2520223980097026,"score_spread":0.13858297038407386,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950788030","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013819852,0.00038493652,0.98042405,0.00017459452,0.00005939781,0.00004698798,0.00010117417,0.0026759368,0.0023130735],"genre_scores_gemma":[0.6201876,0.00045501802,0.37113145,0.00032445148,0.000121696,0.00015778217,0.0005074987,0.00039586655,0.0067186537],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99969304,0.000046793255,0.0000095989135,0.00011881228,0.0000790109,0.000052752406],"domain_scores_gemma":[0.99957794,0.00015370602,0.00004219453,0.00012356589,0.00007190679,0.000030754254],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00053028984,0.0006940028,0.0007257115,0.0003822491,0.00024583712,0.0006632863,0.0017251866,0.0006777895,0.004170785],"category_scores_gemma":[0.001898132,0.00049953145,0.00058630615,0.00041248952,0.0007007599,0.0013578779,0.0017602084,0.0014211973,0.0011534882],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002580916,0.00018975371,0.001,0.00019143065,0.00008405363,0.00016483039,0.00023042833,0.23854053,0.046716973,0.015102982,0.008180457,0.6893404],"study_design_scores_gemma":[0.000007163683,0.00005237473,0.00042222324,0.000007626276,0.000009861823,0.00005384627,0.000016751035,0.9797243,0.007612256,0.010294726,0.0017877854,0.000011054983],"about_ca_topic_score_codex":0.0030894775,"about_ca_topic_score_gemma":0.004201657,"teacher_disagreement_score":0.004170785,"about_ca_system_score_codex":0.0005903066,"about_ca_system_score_gemma":0.00055182393,"threshold_uncertainty_score":0.013952613},"labels":[],"label_agreement":null},{"id":"W2950895666","doi":"","title":"Visual Reasoning with Multi-hop Feature Modulation","year":2018,"lang":"en","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Institute for Advanced Research; Université de Montréal","funders":"","keywords":"Computer science; Hop (telecommunications); Feature (linguistics); Telecommunications","score_opus":0.00956658975243875,"score_gpt":0.25293620468830846,"score_spread":0.24336961493586973,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950895666","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025134934,0.0004786523,0.96336216,0.00040930524,0.00015407597,0.000091170834,0.00026262773,0.002378921,0.007728085],"genre_scores_gemma":[0.7181643,0.0004846432,0.2736675,0.00033680437,0.00014133246,0.00010107803,0.00047653014,0.00021024441,0.006417575],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99956435,0.00008250255,0.000026185558,0.00014803655,0.00012445982,0.00005445192],"domain_scores_gemma":[0.9990482,0.00052307564,0.000063045954,0.00019298574,0.00012103615,0.000051719915],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000492748,0.000716112,0.0006148756,0.0006841521,0.00038433514,0.0017149077,0.0012476791,0.0011051275,0.0101198545],"category_scores_gemma":[0.003947761,0.00028612316,0.00073599146,0.000557495,0.00057797023,0.002543826,0.0013642651,0.0010203852,0.0013632319],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001172584,0.0002722228,0.00082735607,0.000426649,0.00010067673,0.0005534507,0.00024732001,0.06795397,0.075399876,0.044606254,0.011036817,0.79740286],"study_design_scores_gemma":[0.00006586235,0.00012355622,0.0006679752,0.00005178783,0.000052791827,0.00024127819,0.000087256085,0.855185,0.030767336,0.106415294,0.0063134544,0.00002835726],"about_ca_topic_score_codex":0.00219716,"about_ca_topic_score_gemma":0.0015523355,"teacher_disagreement_score":0.0101198545,"about_ca_system_score_codex":0.0005652531,"about_ca_system_score_gemma":0.00035972445,"threshold_uncertainty_score":0.033854306},"labels":[],"label_agreement":null},{"id":"W2952059673","doi":"10.48550/arxiv.1611.07810","title":"A dataset and exploration of models for understanding video data through fill-in-the-blank question-answering","year":2016,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Polytechnique Montréal","funders":"","keywords":"Computer science; Automatic summarization; Benchmark (surveying); Artificial intelligence; Task (project management); Vocabulary; Question answering; Machine learning; Language model; Blank; Field (mathematics); Convolutional neural network; Natural language processing; Information retrieval","score_opus":0.3574155743804664,"score_gpt":0.29138193558292586,"score_spread":0.06603363879754054,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2952059673","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21914122,0.0074028755,0.08619177,0.0051038163,0.00087697257,0.0029264567,0.63604134,0.024515577,0.017800003],"genre_scores_gemma":[0.11220022,0.00089479605,0.11347043,0.00081066176,0.00012475072,0.0012663131,0.76595926,0.0003436677,0.0049298457],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984654,0.00044840315,0.00014916924,0.0005572211,0.00026883234,0.00011097015],"domain_scores_gemma":[0.9970053,0.0014319755,0.0001931396,0.0007185402,0.00045762444,0.00019345575],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017072749,0.0024673091,0.0006708986,0.0019643174,0.00092963077,0.0012759179,0.0030517885,0.0029759705,0.005483332],"category_scores_gemma":[0.009077498,0.00038999927,0.0017859677,0.0020024115,0.00059149944,0.0024453704,0.0013205317,0.0027386097,0.0039054316],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013766082,0.0023466153,0.018799165,0.0036417171,0.0003919051,0.0013772903,0.00088142965,0.045678858,0.013612074,0.006943416,0.63783157,0.26711938],"study_design_scores_gemma":[0.001028572,0.0020963657,0.0478854,0.0010951845,0.00025978612,0.002989604,0.0023854268,0.54036635,0.021615854,0.022205628,0.35776705,0.0003047936],"about_ca_topic_score_codex":0.028797261,"about_ca_topic_score_gemma":0.06400169,"teacher_disagreement_score":0.028797261,"about_ca_system_score_codex":0.0021051152,"about_ca_system_score_gemma":0.0013211487,"threshold_uncertainty_score":0.05725932},"labels":[],"label_agreement":null},{"id":"W2952769376","doi":"10.48550/arxiv.1806.07011","title":"VirtualHome: Simulating Household Activities via Programs","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; Defense Advanced Research Projects Agency; Intelligence Advanced Research Projects Activity; Samsung; Nvidia","keywords":"Computer science; Task (project management); Variety (cybernetics); Representation (politics); Human–computer interaction; Interface (matter); Code (set theory); Game engine; Artificial intelligence; Programming language","score_opus":0.0827059822227267,"score_gpt":0.21093085153611998,"score_spread":0.12822486931339327,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2952769376","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.42291117,0.00030221936,0.5511867,0.0003572113,0.00009070214,0.0005008319,0.0052141524,0.010574449,0.008862592],"genre_scores_gemma":[0.81032157,0.000226662,0.17984593,0.00011850325,0.0000111135905,0.0005235292,0.004983422,0.0005584905,0.003410816],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998242,0.000069403955,0.000008035065,0.00005224601,0.000026910848,0.000019222467],"domain_scores_gemma":[0.99972206,0.00017301756,0.000019331224,0.000037025042,0.000021002585,0.000027604929],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00022217516,0.00071164634,0.0002695321,0.0003773669,0.00019407745,0.0005582976,0.0012052424,0.00067886716,0.003382182],"category_scores_gemma":[0.001262137,0.00026152746,0.00052128587,0.0002798359,0.00060405844,0.00083581475,0.0008710161,0.00059898495,0.00045652234],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000546777,0.0004374281,0.008295019,0.00043071108,0.00008616341,0.00053667364,0.0010324277,0.8956008,0.0135818375,0.012029829,0.00888815,0.058534153],"study_design_scores_gemma":[0.000042490727,0.000083881365,0.0013598227,0.000022753064,0.000012703445,0.000080675076,0.000164585,0.9821862,0.0053297095,0.004042203,0.0066580945,0.000016830003],"about_ca_topic_score_codex":0.008723053,"about_ca_topic_score_gemma":0.012283568,"teacher_disagreement_score":0.008723053,"about_ca_system_score_codex":0.00046849975,"about_ca_system_score_gemma":0.00040102168,"threshold_uncertainty_score":0.017344534},"labels":[],"label_agreement":null},{"id":"W2960202457","doi":"10.1145/3306346.3322941","title":"PlanIT","year":2019,"lang":"en","type":"article","venue":"ACM Transactions on Graphics","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":207,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"FP7 Ideas: European Research Council; National Science Foundation","keywords":"Computer science; Scene graph; Graph; Relation (database); Generative model; Artificial intelligence; Spatial relation; Representation (politics); Theoretical computer science; Set (abstract data type); Generative grammar; Data mining; Rendering (computer graphics); Programming language","score_opus":0.01302205938873805,"score_gpt":0.2537439968164512,"score_spread":0.24072193742771317,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2960202457","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0015873673,0.00037650118,0.9102837,0.00068884395,0.00028282692,0.00024189045,0.003124317,0.03296761,0.050446935],"genre_scores_gemma":[0.06825519,0.0007673282,0.8454546,0.0009434745,0.00016353418,0.00074248266,0.013144113,0.009237201,0.061292134],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99919254,0.00013747715,0.000047098052,0.00023669396,0.0002819533,0.00010432552],"domain_scores_gemma":[0.9995198,0.000111937465,0.000033487508,0.00018944211,0.00010257765,0.000042694795],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007882133,0.0014997006,0.0007003132,0.0011297879,0.0009505751,0.00339281,0.0031975189,0.0013828245,0.07979538],"category_scores_gemma":[0.0022251988,0.00081884325,0.0019732756,0.0008244314,0.0011095882,0.0039934427,0.0034949721,0.0020282122,0.025732333],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002513933,0.00011709476,0.00073737884,0.0006535036,0.00007042472,0.000341397,0.00029877204,0.044158287,0.00362993,0.46410054,0.13048936,0.35515192],"study_design_scores_gemma":[0.00007125749,0.00008179215,0.00025673147,0.00014460011,0.000041174655,0.0003632831,0.000085305284,0.2670994,0.0073300847,0.227057,0.49741963,0.000049706858],"about_ca_topic_score_codex":0.0059121842,"about_ca_topic_score_gemma":0.012142273,"teacher_disagreement_score":0.07979538,"about_ca_system_score_codex":0.0015867815,"about_ca_system_score_gemma":0.001819408,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2962849532","doi":"","title":"Teaching Machines to Describe Images with Natural Language Feedback","year":2017,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Closed captioning; Computer science; Natural language; Focus (optics); Artificial intelligence; Sentence; Point (geometry); Natural (archaeology); Quality (philosophy); Phrase; Natural language processing; SIGNAL (programming language); Robot; Human–computer interaction; Machine learning; Image (mathematics)","score_opus":0.010935333915803007,"score_gpt":0.28775438830282307,"score_spread":0.2768190543870201,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2962849532","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03316235,0.00040769958,0.9558847,0.0013857494,0.00015673925,0.00011824261,0.00013484324,0.003208072,0.0055416767],"genre_scores_gemma":[0.5895531,0.00081686594,0.3983003,0.00082814554,0.0001715387,0.0002684662,0.00039717671,0.0003492072,0.00931522],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99956316,0.00021545636,0.00001688359,0.00010637766,0.0000734343,0.000024659555],"domain_scores_gemma":[0.99686927,0.002179979,0.000252858,0.00033720685,0.00028801573,0.00007265253],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010817766,0.00093918113,0.00036995273,0.00026112833,0.00021145366,0.00073901744,0.0012010708,0.0012222962,0.0034862163],"category_scores_gemma":[0.009320332,0.00036360585,0.0003864178,0.000259935,0.0009971771,0.0021220127,0.0008336997,0.0017695659,0.0013672437],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021608744,0.00034629708,0.0026650596,0.00069885555,0.00009992635,0.00028657148,0.0012357244,0.38357168,0.06168546,0.05068436,0.018801842,0.47970808],"study_design_scores_gemma":[0.000023136967,0.00010526937,0.00035187407,0.000040193157,0.00001824109,0.00010526044,0.000082172985,0.94119495,0.017041128,0.0330693,0.007941335,0.000027135195],"about_ca_topic_score_codex":0.0012655214,"about_ca_topic_score_gemma":0.0025575722,"teacher_disagreement_score":0.0034862163,"about_ca_system_score_codex":0.0004810489,"about_ca_system_score_gemma":0.0004600273,"threshold_uncertainty_score":0.011662543},"labels":[],"label_agreement":null},{"id":"W2962860923","doi":"10.1109/iccv.2019.00285","title":"Lifelong GAN: Continual Learning for Conditional Image Generation","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Forgetting; Lifelong learning; Computer science; Artificial intelligence; Artificial neural network; Generative grammar; Machine learning; Image (mathematics); Task (project management); Generative model; Cognitive psychology; Engineering; Psychology","score_opus":0.025556129044986057,"score_gpt":0.31175500620246316,"score_spread":0.2861988771574771,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2962860923","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018841306,0.00028869204,0.97555906,0.00025495188,0.000048671325,0.00006417976,0.00017293522,0.0024636418,0.002306488],"genre_scores_gemma":[0.67882544,0.00023333398,0.31169227,0.0005095484,0.00007903801,0.0002651028,0.0009955146,0.0006172019,0.0067825266],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999622,0.00012256899,0.00001308051,0.000118077995,0.000079649304,0.00004460335],"domain_scores_gemma":[0.9986985,0.00063691946,0.00008137367,0.00036254738,0.00014174574,0.000078854944],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012453212,0.0008899863,0.00056774926,0.00028379026,0.00027564995,0.0006726435,0.0020093273,0.0011887703,0.0036958277],"category_scores_gemma":[0.003311648,0.00042891872,0.0004968878,0.0002858058,0.0009721043,0.0016210829,0.0015920443,0.002244388,0.00082837],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020344299,0.00017308202,0.0015635113,0.00014164788,0.000067136505,0.00022059485,0.00015536604,0.77870065,0.011655477,0.029488089,0.009194057,0.16843694],"study_design_scores_gemma":[0.000005530158,0.000016889198,0.00005954832,0.0000045254324,0.0000023862617,0.00002436516,0.0000035953283,0.99068636,0.0017014304,0.0069748033,0.00051658135,0.0000039544957],"about_ca_topic_score_codex":0.0021590167,"about_ca_topic_score_gemma":0.004327985,"teacher_disagreement_score":0.0036958277,"about_ca_system_score_codex":0.00084323395,"about_ca_system_score_gemma":0.00062496774,"threshold_uncertainty_score":0.012363732},"labels":[],"label_agreement":null},{"id":"W2962879739","doi":"10.1007/978-3-319-18356-5_22","title":"Learning Paired-Associate Images with an Unsupervised Deep Learning Architecture","year":2015,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Acadia University","funders":"","keywords":"Computer science; MNIST database; Artificial intelligence; Associative property; Restricted Boltzmann machine; Content-addressable memory; Unsupervised learning; Deep learning; Pattern recognition (psychology); Representation (politics); Boltzmann machine; Modal; Artificial neural network; Machine learning; Mathematics","score_opus":0.01598576285133331,"score_gpt":0.25674946916856856,"score_spread":0.24076370631723526,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2962879739","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01394954,0.00024819555,0.9795065,0.000176113,0.00013522834,0.000040581,0.00013855829,0.0017799459,0.004025314],"genre_scores_gemma":[0.23710068,0.0005099287,0.74521536,0.00042381184,0.0001475702,0.000106340914,0.00083556987,0.00034798583,0.015312779],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99961424,0.00006053837,0.000013044209,0.00015858337,0.00009733128,0.00005626257],"domain_scores_gemma":[0.99952114,0.00011617645,0.000036371857,0.00017817113,0.000099390985,0.00004878589],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00062622223,0.00093515095,0.0009009595,0.0005553394,0.00033737568,0.0011566018,0.0021287461,0.0016105497,0.0083500985],"category_scores_gemma":[0.0019920564,0.00064071076,0.0009200682,0.0009463383,0.00080641586,0.0024113136,0.0025364454,0.0025654668,0.003499103],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029229335,0.00018785901,0.0008234491,0.00012606486,0.000100253885,0.00011780259,0.00006971889,0.060165845,0.028715964,0.021415703,0.009762454,0.87822264],"study_design_scores_gemma":[0.000019702777,0.00011376541,0.00032194867,0.000030721596,0.00003438802,0.00019567236,0.0000310443,0.9379443,0.0215431,0.034742143,0.0050006467,0.000022534758],"about_ca_topic_score_codex":0.0013794072,"about_ca_topic_score_gemma":0.0024916087,"teacher_disagreement_score":0.0083500985,"about_ca_system_score_codex":0.0004132901,"about_ca_system_score_gemma":0.00067033677,"threshold_uncertainty_score":0.027933896},"labels":[],"label_agreement":null},{"id":"W2963062932","doi":"10.1609/aaai.v30i1.10475","title":"SentiCap: Generating Image Descriptions with Sentiments","year":2016,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":214,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Australian Research Council; Instituto de Ciencias del Mar y Limnología, Universidad Nacional Autónoma de México; Australian Government; Institute for Catastrophic Loss Reduction; Nvidia","keywords":"Closed captioning; Computer science; Natural language processing; Stylized fact; Artificial intelligence; Image (mathematics); Vocabulary; Sentiment analysis; Regularization (linguistics); Quality (philosophy); Linguistics","score_opus":0.05085603367949731,"score_gpt":0.29375943901469953,"score_spread":0.24290340533520222,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963062932","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06870779,0.0013044303,0.77285355,0.000977938,0.0009836159,0.0014596622,0.012872718,0.11979798,0.021042356],"genre_scores_gemma":[0.28951946,0.000801025,0.6551006,0.00061843544,0.00030382484,0.0012256484,0.031605527,0.005731046,0.01509449],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99944645,0.00017730422,0.00003400127,0.000121554265,0.00018412631,0.00003670325],"domain_scores_gemma":[0.9985885,0.0005369618,0.00010567865,0.00025086413,0.00045516388,0.00006274948],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001048335,0.0018448462,0.0004621265,0.0014673553,0.0004114996,0.0010299069,0.0013134918,0.0012080185,0.010776045],"category_scores_gemma":[0.004670191,0.00048797805,0.0010834098,0.0007523642,0.00042927722,0.0015826494,0.0012547754,0.0010296734,0.0043313005],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009490997,0.00048153935,0.0026873713,0.0021589485,0.00037130542,0.0010922373,0.0011746145,0.058912404,0.10761227,0.011253965,0.22802015,0.58528614],"study_design_scores_gemma":[0.00008880949,0.00031307453,0.0019413675,0.00009349947,0.000077149896,0.00041365466,0.00026990435,0.86598986,0.058908515,0.011883581,0.059922546,0.00009804838],"about_ca_topic_score_codex":0.0018368092,"about_ca_topic_score_gemma":0.0025554406,"teacher_disagreement_score":0.010776045,"about_ca_system_score_codex":0.000664828,"about_ca_system_score_gemma":0.00036674005,"threshold_uncertainty_score":0.036049426},"labels":[],"label_agreement":null},{"id":"W2963202404","doi":"","title":"FigureQA: An Annotated Figure Dataset for Visual Reasoning","year":2017,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Polytechnique Montréal","funders":"","keywords":"Computer science; Plot (graphics); Visual reasoning; Artificial intelligence; Task (project management); Intersection (aeronautics); Bar chart; Scatter plot; Bounding overwatch; Minimum bounding box; Natural language processing; Smoothness; Baseline (sea); Machine learning; Line (geometry); Image (mathematics); Mathematics","score_opus":0.07102971897405239,"score_gpt":0.44039492234294464,"score_spread":0.3693652033688922,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963202404","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020272039,0.0035670933,0.027735077,0.0015299884,0.00053711794,0.0009825759,0.88375705,0.039788935,0.02183005],"genre_scores_gemma":[0.03110679,0.0005694341,0.049956262,0.0005693065,0.0000579706,0.00084085425,0.9091921,0.0016312818,0.0060759075],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985121,0.0003153567,0.00016310561,0.0004886497,0.0004116543,0.0001092068],"domain_scores_gemma":[0.9968162,0.0015438223,0.000179866,0.0006713833,0.00058972865,0.00019890864],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000887662,0.002859113,0.00086360914,0.00413312,0.000972986,0.0020639729,0.0035360325,0.003981805,0.052810386],"category_scores_gemma":[0.009019935,0.0006298453,0.0018177312,0.0032333082,0.0008155882,0.0031556725,0.002465477,0.0021736524,0.027491922],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044934495,0.00027775823,0.0029234488,0.003242444,0.00011587037,0.0009652507,0.00052553747,0.0043058144,0.004877331,0.0054303203,0.91365856,0.0632283],"study_design_scores_gemma":[0.0004486044,0.00018594164,0.010878204,0.0008809539,0.00008508272,0.0013777225,0.0011094688,0.042451795,0.0075393994,0.015679345,0.9192148,0.0001486458],"about_ca_topic_score_codex":0.023210132,"about_ca_topic_score_gemma":0.049342487,"teacher_disagreement_score":0.052810386,"about_ca_system_score_codex":0.0020678835,"about_ca_system_score_gemma":0.0012806132,"threshold_uncertainty_score":0.17666835},"labels":[],"label_agreement":null},{"id":"W2963245493","doi":"","title":"Modulating early visual processing by language","year":2017,"lang":"en","type":"article","venue":"LillOA (Université de Lille (University Of Lille))","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":147,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Canadian Institute for Advanced Research; Nvidia","keywords":"Computer science; Natural language processing; Visual language; Artificial intelligence; Linguistics; Philosophy","score_opus":0.004933358070285787,"score_gpt":0.21979079856117845,"score_spread":0.21485744049089267,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963245493","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5801118,0.016980978,0.06481376,0.011775308,0.006034236,0.0006649269,0.0037490278,0.003970833,0.3118991],"genre_scores_gemma":[0.8383726,0.006031399,0.017081179,0.0054677925,0.0016762076,0.00046636764,0.0015919721,0.001622778,0.1276896],"study_design_codex":"bench_or_experimental","study_design_gemma":"not_applicable","domain_scores_codex":[0.99975985,0.000028668626,0.000011851528,0.00009284915,0.000046601275,0.00006026491],"domain_scores_gemma":[0.999015,0.00047466528,0.000106669395,0.00008840083,0.00014019114,0.00017506405],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00046457292,0.0011811097,0.00070302375,0.0003601743,0.0004101328,0.0024149248,0.0007930335,0.0012699692,0.07735024],"category_scores_gemma":[0.002727338,0.00024120358,0.0004826926,0.00024089905,0.00039360128,0.0013582193,0.0008019154,0.0010883948,0.0099759335],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00076599885,0.00008401967,0.0004106471,0.0003453602,0.000025836292,0.00015755193,0.00010897598,0.00027051027,0.9253914,0.0026647174,0.005766234,0.06400882],"study_design_scores_gemma":[0.0012995279,0.0029198446,0.12146187,0.0010394261,0.00086653087,0.0017718606,0.0010842741,0.017139265,0.6432858,0.050867617,0.15802549,0.00023849617],"about_ca_topic_score_codex":0.0015601559,"about_ca_topic_score_gemma":0.0011234804,"teacher_disagreement_score":0.07735024,"about_ca_system_score_codex":0.0005581912,"about_ca_system_score_gemma":0.0005914543,"threshold_uncertainty_score":0.2587623},"labels":[],"label_agreement":null},{"id":"W2963299217","doi":"","title":"A Neural Compositional Paradigm for Image Captioning","year":2018,"lang":"en","type":"article","venue":"The HKU Scholars Hub (University of Hong Kong)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Closed captioning; Computer science; Semantics (computer science); Generalization; Natural language processing; Artificial intelligence; Syntax; Representation (politics); Image (mathematics); Programming language; Mathematics","score_opus":0.01437617044879723,"score_gpt":0.2468396403155849,"score_spread":0.23246346986678768,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963299217","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0038683848,0.00018742179,0.9916746,0.00016142523,0.000062424886,0.000068120324,0.00006879564,0.0006008729,0.0033079702],"genre_scores_gemma":[0.25409308,0.00064875686,0.73599833,0.0003939862,0.00022339802,0.00042879535,0.0006164403,0.00026063572,0.00733651],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995209,0.00014815228,0.000026242344,0.0001796048,0.00009003973,0.000035049154],"domain_scores_gemma":[0.9994604,0.00022210945,0.0000435949,0.00012464274,0.00011803167,0.00003119735],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009572725,0.0008981592,0.00047540208,0.00072842446,0.00063295313,0.0009322952,0.0017627011,0.0013577465,0.0045088124],"category_scores_gemma":[0.0030086848,0.00040617876,0.0011307434,0.0007595345,0.00122641,0.0024473858,0.0014220177,0.0017409291,0.0011757112],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019572875,0.00017348131,0.0004142384,0.0005127179,0.00011110885,0.00033179708,0.00057837035,0.17661317,0.061131075,0.19258264,0.010562429,0.5567932],"study_design_scores_gemma":[0.00001716032,0.0000786664,0.0001490277,0.00002603859,0.000028867506,0.00016003258,0.000037884583,0.8932787,0.01128026,0.08661977,0.008299908,0.00002368951],"about_ca_topic_score_codex":0.0016690246,"about_ca_topic_score_gemma":0.0025505694,"teacher_disagreement_score":0.0045088124,"about_ca_system_score_codex":0.0006930669,"about_ca_system_score_gemma":0.0007149016,"threshold_uncertainty_score":0.015083492},"labels":[],"label_agreement":null},{"id":"W2963398599","doi":"10.1109/cvpr.2016.500","title":"Ask Me Anything: Free-Form Visual Question Answering Based on Knowledge from External Sources","year":2016,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":368,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Question answering; Computer science; Image (mathematics); Information retrieval; Knowledge base; Ask price; Representation (politics); Artificial intelligence; Knowledge representation and reasoning; Natural language; Questions and answers; Natural language processing","score_opus":0.015678643470038032,"score_gpt":0.30608828207007777,"score_spread":0.2904096386000397,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963398599","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027665164,0.0013280016,0.94493824,0.0015306129,0.00013819069,0.0005558711,0.0024722577,0.012318505,0.009053255],"genre_scores_gemma":[0.43024987,0.00052593835,0.54694605,0.0012331395,0.00023671697,0.00083507976,0.009857893,0.0005380867,0.009577236],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981681,0.0006392327,0.000092384835,0.00063698407,0.00032495343,0.00013836002],"domain_scores_gemma":[0.9964898,0.0021689476,0.00020854408,0.00055986596,0.0004203219,0.00015247286],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020593402,0.0017797507,0.001035149,0.0019407618,0.000702317,0.0020415792,0.003509652,0.0031743976,0.008840864],"category_scores_gemma":[0.009059206,0.0005754305,0.0016830579,0.0009942789,0.0012384859,0.0060286084,0.0031702905,0.0023569164,0.0027425142],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008793915,0.0009025443,0.0043165083,0.001992815,0.0003212105,0.00081157044,0.0031685382,0.052233815,0.040232413,0.042639684,0.057939965,0.7945615],"study_design_scores_gemma":[0.00014808262,0.0002351719,0.0021055106,0.00016266224,0.00017542216,0.00042684254,0.00062763214,0.83180285,0.021907384,0.112586625,0.029724274,0.00009748845],"about_ca_topic_score_codex":0.008442337,"about_ca_topic_score_gemma":0.008949405,"teacher_disagreement_score":0.008840864,"about_ca_system_score_codex":0.0013133042,"about_ca_system_score_gemma":0.0010805731,"threshold_uncertainty_score":0.029575646},"labels":[],"label_agreement":null},{"id":"W2963499204","doi":"","title":"VSE++: Improving Visual-Semantic Embeddings with Hard Negatives.","year":2018,"lang":"en","type":"article","venue":"British Machine Vision Conference","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":230,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Modal; Ranking (information retrieval); Information retrieval; Learning to rank; Artificial intelligence; Simple (philosophy); Image retrieval; Image (mathematics); Machine learning; Data mining; Pattern recognition (psychology)","score_opus":0.010031982784397351,"score_gpt":0.28159681845750656,"score_spread":0.2715648356731092,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963499204","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.061146248,0.0021657306,0.9096893,0.0005135356,0.00056134857,0.00034063522,0.0025958598,0.015018269,0.007969151],"genre_scores_gemma":[0.45823282,0.0007878417,0.49939704,0.0009950993,0.00037440882,0.0005242042,0.016672939,0.001641488,0.02137415],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99837434,0.0004745629,0.000100996156,0.00035461065,0.0005726981,0.00012272134],"domain_scores_gemma":[0.9975063,0.0007613799,0.00019137246,0.00085548015,0.00059205294,0.000093393515],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020193022,0.0021549617,0.00135451,0.0017516011,0.0005357601,0.0015885753,0.002230057,0.0015561722,0.0050295424],"category_scores_gemma":[0.011038935,0.00044735958,0.00092766376,0.0014122853,0.0009563417,0.0046848273,0.0032593748,0.0019618622,0.004707535],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00072235893,0.00077827653,0.0030520912,0.00056963443,0.00019136017,0.00019628131,0.00018234986,0.039947905,0.027220504,0.014280233,0.05327483,0.85958415],"study_design_scores_gemma":[0.000110179804,0.00065392774,0.0012618228,0.0000782761,0.00007711967,0.0005975178,0.00024192089,0.9062929,0.028281745,0.03556922,0.026759418,0.000075867945],"about_ca_topic_score_codex":0.0018672397,"about_ca_topic_score_gemma":0.0030033875,"teacher_disagreement_score":0.0050295424,"about_ca_system_score_codex":0.0005045528,"about_ca_system_score_gemma":0.00072101806,"threshold_uncertainty_score":0.016825438},"labels":[],"label_agreement":null},{"id":"W2963651499","doi":"10.48550/arxiv.1707.03017","title":"Learning Visual Reasoning Without Strong Priors","year":2017,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Artificial intelligence; Visual reasoning; Exploit; Machine learning; Normalization (sociology); Benchmark (surveying); Process (computing); Deep learning; Natural language processing","score_opus":0.013999791537374134,"score_gpt":0.282005870834834,"score_spread":0.2680060792974599,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963651499","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02762683,0.0005462788,0.9527995,0.00094170374,0.00011294956,0.000107334614,0.0006687,0.010514421,0.0066823564],"genre_scores_gemma":[0.5369622,0.0004170564,0.44456175,0.0012116662,0.00013689372,0.0001665084,0.0031462838,0.000695593,0.012702147],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99893564,0.00020262763,0.000047303427,0.000536175,0.00017346609,0.00010479412],"domain_scores_gemma":[0.9977623,0.0010913672,0.00016686598,0.0006167184,0.00024762578,0.00011502465],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001365502,0.0014590438,0.00075425365,0.00070391205,0.00045471193,0.0021909187,0.002684669,0.0021802534,0.00823248],"category_scores_gemma":[0.0078065163,0.0007529215,0.0020740032,0.00040437467,0.0014710566,0.004807952,0.0022011246,0.0037445754,0.0030976804],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00053216255,0.0003065433,0.0018543934,0.00063955743,0.00020587664,0.00032189066,0.0003848792,0.20986328,0.035474494,0.06596499,0.031031689,0.6534202],"study_design_scores_gemma":[0.000032586384,0.000041135187,0.00040145288,0.000040483235,0.00002733022,0.00007347588,0.00004293526,0.87522644,0.0106019005,0.10964776,0.003844829,0.000019597519],"about_ca_topic_score_codex":0.0047879815,"about_ca_topic_score_gemma":0.007436576,"teacher_disagreement_score":0.00823248,"about_ca_system_score_codex":0.0013292491,"about_ca_system_score_gemma":0.0012412888,"threshold_uncertainty_score":0.027540386},"labels":[],"label_agreement":null},{"id":"W2963742410","doi":"","title":"DOM-Q-NET: Grounded RL on Structured Language","year":2019,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Reinforcement learning; Artificial intelligence; Task (project management); Machine learning; The Internet; String (physics); Graph; Natural language processing; Theoretical computer science; World Wide Web","score_opus":0.021628009855171406,"score_gpt":0.34219214213863863,"score_spread":0.32056413228346725,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963742410","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014629272,0.00013257562,0.9782696,0.00026018062,0.00006742907,0.000052885825,0.00016773805,0.0030707903,0.0033495969],"genre_scores_gemma":[0.6955812,0.00020482749,0.29443246,0.00051852816,0.000042239768,0.00026955744,0.0006407877,0.00052634935,0.007784131],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99971384,0.000104587794,0.000012128828,0.00007521633,0.00005296361,0.000041236064],"domain_scores_gemma":[0.9992791,0.0004417508,0.000038579383,0.000089193825,0.00009337445,0.00005802536],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008564518,0.0008285326,0.0006302873,0.00031090205,0.00038841955,0.00086991553,0.0017078853,0.0012527251,0.004470039],"category_scores_gemma":[0.0032929466,0.00056009623,0.00053745025,0.00027193362,0.001049613,0.0015870446,0.0015639263,0.0018369752,0.0011255909],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013008129,0.0000602173,0.00045259367,0.00005113189,0.000024452622,0.00008480411,0.00005772074,0.93478155,0.0019225556,0.0145398555,0.002216841,0.045678146],"study_design_scores_gemma":[0.000007202314,0.000009002368,0.000015689498,0.000003177802,0.0000014823759,0.0000033603033,0.0000022793006,0.9940783,0.00028291627,0.0053361338,0.00025855514,0.0000019559245],"about_ca_topic_score_codex":0.00942892,"about_ca_topic_score_gemma":0.013238012,"teacher_disagreement_score":0.00942892,"about_ca_system_score_codex":0.0009876103,"about_ca_system_score_gemma":0.0012570017,"threshold_uncertainty_score":0.018748045},"labels":[],"label_agreement":null},{"id":"W2964028737","doi":"10.18653/v1/w17-2629","title":"Adversarial Generation of Natural Language","year":2017,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":115,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Institute for Advanced Research; Université de Montréal; Polytechnique Montréal","funders":"Nvidia","keywords":"Natural language generation; Computer science; Adversarial system; Natural language; Context (archaeology); Artificial intelligence; Generative grammar; Probabilistic logic; Natural language understanding; Natural language processing; Rule-based machine translation; Sentence; Language model; Estimator; Theoretical computer science; Mathematics","score_opus":0.02246753577189608,"score_gpt":0.31643478305866657,"score_spread":0.2939672472867705,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964028737","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05130355,0.00036865607,0.93744147,0.00063517195,0.00017844664,0.0001983738,0.0008069893,0.0018669544,0.007200435],"genre_scores_gemma":[0.7993308,0.0003201344,0.18580799,0.0006291973,0.0001041514,0.00060665666,0.0026598086,0.00047539253,0.010065948],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99934846,0.0002887538,0.000020132393,0.00015617438,0.00013346421,0.00005298921],"domain_scores_gemma":[0.99782676,0.0015990916,0.0000952229,0.00025576414,0.00017151011,0.000051561936],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011594058,0.00079103495,0.00046551717,0.0003974449,0.00026976038,0.000473501,0.0009285892,0.00069998065,0.0039490703],"category_scores_gemma":[0.0056466004,0.00023489563,0.0006066897,0.0003346239,0.00080890866,0.0007808216,0.0011095987,0.0013836602,0.0008216387],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015895767,0.000102733706,0.0011092167,0.00016499341,0.000058828988,0.00041351645,0.00021004416,0.8357794,0.010287289,0.06765134,0.012091698,0.0719719],"study_design_scores_gemma":[0.000011421568,0.000020597856,0.00009180346,0.000009957861,0.000005149317,0.000046578778,0.0000134017855,0.97446126,0.0020857113,0.021488812,0.0017579856,0.000007205822],"about_ca_topic_score_codex":0.000974897,"about_ca_topic_score_gemma":0.0015726846,"teacher_disagreement_score":0.0039490703,"about_ca_system_score_codex":0.0005327277,"about_ca_system_score_gemma":0.00057123124,"threshold_uncertainty_score":0.013210952},"labels":[],"label_agreement":null},{"id":"W2964248098","doi":"10.1007/978-3-030-11018-5_21","title":"Learnable Pooling Methods for Video Classification","year":2019,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Pooling; Computer science; Artificial intelligence","score_opus":0.03735968409245802,"score_gpt":0.35151771802476095,"score_spread":0.3141580339323029,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964248098","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0034569926,0.0021664866,0.9913084,0.00014836491,0.00010990292,0.0000314436,0.0001880767,0.0011067573,0.0014835855],"genre_scores_gemma":[0.13664421,0.0032880015,0.8270286,0.0003402202,0.0006370082,0.00024294946,0.0022746103,0.00074606173,0.028798234],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994779,0.00010296853,0.00003462295,0.0001656956,0.0001385803,0.000080152065],"domain_scores_gemma":[0.9994771,0.00020869264,0.00004409316,0.00012616428,0.0001141875,0.000029809045],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012177634,0.0016336823,0.0014318635,0.0010067425,0.00038644692,0.0013084791,0.0023898478,0.0013906255,0.0066604204],"category_scores_gemma":[0.0022566156,0.00055667036,0.0012996162,0.0017271143,0.00057735824,0.0025488336,0.0018076744,0.0019205988,0.0026497345],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013670172,0.00009046797,0.00025368328,0.00018707004,0.00014455637,0.00005053707,0.00005129616,0.043359112,0.013179116,0.016956074,0.013248771,0.9123425],"study_design_scores_gemma":[0.000014412645,0.000060169008,0.0004013362,0.000032890974,0.000055570654,0.00008046321,0.000020844302,0.9496359,0.010152756,0.03274731,0.0067788423,0.0000193886],"about_ca_topic_score_codex":0.0049878135,"about_ca_topic_score_gemma":0.005705515,"teacher_disagreement_score":0.0066604204,"about_ca_system_score_codex":0.00090582925,"about_ca_system_score_gemma":0.0006351292,"threshold_uncertainty_score":0.022281349},"labels":[],"label_agreement":null},{"id":"W2964516821","doi":"","title":"TRECVID 2018: Benchmarking Video Activity Detection, Video Captioning and Matching, Video Storytelling Linking and Video Search","year":2018,"lang":"en","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"BC Research (Canada)","funders":"","keywords":"Computer science; Closed captioning; Benchmarking; Storytelling; Matching (statistics); Videoconferencing; Artificial intelligence; Video tracking; Computer vision; Video processing; Multimedia; Narrative; Image (mathematics); Mathematics","score_opus":0.012794704856313892,"score_gpt":0.24439575751913414,"score_spread":0.23160105266282024,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964516821","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3619009,0.053805765,0.054704234,0.0035833342,0.0127517395,0.007102714,0.28664827,0.18416809,0.03533494],"genre_scores_gemma":[0.1601061,0.0041747564,0.082670294,0.0010752585,0.0011821709,0.0021031639,0.72296965,0.0040230723,0.021695666],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98588276,0.0029498225,0.0013219814,0.004179716,0.004251128,0.0014145806],"domain_scores_gemma":[0.99085546,0.003225136,0.0004820027,0.0016123643,0.0025711823,0.0012537852],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012277131,0.011510814,0.006481943,0.013512826,0.0027470735,0.004373526,0.010717807,0.00738135,0.0127248885],"category_scores_gemma":[0.017440196,0.0018427296,0.0046103443,0.008248247,0.0021163637,0.0076636258,0.0052302605,0.004814701,0.017647067],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0052659875,0.003295894,0.0043999013,0.00520521,0.0024398246,0.0007979994,0.00023669376,0.022164185,0.015606188,0.0013219758,0.6114907,0.32777545],"study_design_scores_gemma":[0.0049341945,0.0047074757,0.042642437,0.0010144508,0.0030196274,0.004656139,0.0015760522,0.70192784,0.07890203,0.0059523648,0.15009338,0.0005740249],"about_ca_topic_score_codex":0.09167215,"about_ca_topic_score_gemma":0.08296578,"teacher_disagreement_score":0.09167215,"about_ca_system_score_codex":0.005778232,"about_ca_system_score_gemma":0.004580932,"threshold_uncertainty_score":0.18227714},"labels":[],"label_agreement":null},{"id":"W2968104955","doi":"10.1109/cvpr.2019.00675","title":"Streamlined Dense Video Captioning","year":2019,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":153,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Parks and Wilderness Society","funders":"","keywords":"Closed captioning; Computer science; Event (particle physics); Context (archaeology); Dependency (UML); Task (project management); Artificial intelligence; Storytelling; Sequence (biology); Natural language processing; Image (mathematics); Narrative; Linguistics","score_opus":0.006586964800575043,"score_gpt":0.24543348210477273,"score_spread":0.23884651730419768,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2968104955","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037485808,0.0037710853,0.90888107,0.0008657839,0.0010230655,0.0010985086,0.007983232,0.029375762,0.009515652],"genre_scores_gemma":[0.28887093,0.002161966,0.660171,0.000781469,0.0007245482,0.0008278258,0.03130648,0.0015495163,0.01360627],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99895513,0.0002535766,0.000051499923,0.0003949173,0.00025033808,0.00009457771],"domain_scores_gemma":[0.9983907,0.00058033376,0.00012587734,0.00035462054,0.00045077808,0.00009778332],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010797919,0.0027171774,0.0011609929,0.001611401,0.00057487935,0.0013146669,0.002272602,0.0017110165,0.007027224],"category_scores_gemma":[0.0047604633,0.0005576361,0.00092688366,0.0015359113,0.0006385685,0.002474082,0.0017876219,0.0019672867,0.0031880036],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007548813,0.00027731957,0.00096916384,0.0008592517,0.00014198659,0.00080234365,0.0003685034,0.11489739,0.047914084,0.004978505,0.09746293,0.7305737],"study_design_scores_gemma":[0.000080261816,0.00025740915,0.00068943633,0.00006961274,0.000055621407,0.00046508916,0.00012752194,0.92552584,0.03556396,0.0075266412,0.029583694,0.0000550036],"about_ca_topic_score_codex":0.00466858,"about_ca_topic_score_gemma":0.0075946664,"teacher_disagreement_score":0.007027224,"about_ca_system_score_codex":0.0009278184,"about_ca_system_score_gemma":0.00087874883,"threshold_uncertainty_score":0.02350843},"labels":[],"label_agreement":null},{"id":"W2969127500","doi":"10.1109/cvpr.2019.00853","title":"Progressive Attention Memory Network for Movie Story Question Answering","year":2019,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":90,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Question answering; Computer science; Inference; Modality (human–computer interaction); Benchmark (surveying); Artificial intelligence; Fusion mechanism; Feature (linguistics); Natural language processing; Scheme (mathematics); Information retrieval; Fusion; Linguistics","score_opus":0.008055523171589318,"score_gpt":0.27657397358735764,"score_spread":0.26851845041576833,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2969127500","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12954602,0.0049068597,0.83603865,0.002084441,0.00038687018,0.0004972578,0.0021866078,0.008420245,0.015933022],"genre_scores_gemma":[0.81353647,0.00088338344,0.16799252,0.0011257093,0.00025441212,0.00034239286,0.0031885228,0.00012469068,0.01255185],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996971,0.000067247434,0.0000146857865,0.00012585754,0.000050627154,0.000044477056],"domain_scores_gemma":[0.9994942,0.00026320954,0.000037212052,0.00006533535,0.00010093679,0.000039102393],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00075366057,0.0008669474,0.0005761242,0.00085411314,0.0005101176,0.0006592099,0.0018667385,0.0012365197,0.006650894],"category_scores_gemma":[0.0028856597,0.0002633696,0.0005717921,0.00064837415,0.00046018546,0.0022583543,0.0012415602,0.0011612175,0.0010453609],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078945555,0.0003687118,0.0033340529,0.0004282726,0.00013319343,0.0002764698,0.00048589238,0.07133485,0.018244231,0.016004462,0.024772288,0.8638282],"study_design_scores_gemma":[0.000040589228,0.00018406277,0.0012398055,0.000036289137,0.00008955262,0.00014860662,0.00011214722,0.9494387,0.0082532,0.032208964,0.008220079,0.000027951854],"about_ca_topic_score_codex":0.011148907,"about_ca_topic_score_gemma":0.016033735,"teacher_disagreement_score":0.011148907,"about_ca_system_score_codex":0.0010105078,"about_ca_system_score_gemma":0.0008527113,"threshold_uncertainty_score":0.02224946},"labels":[],"label_agreement":null},{"id":"W2973802306","doi":"10.1109/iccv.2019.00900","title":"Watch, Listen and Tell: Multi-Modal Weakly Supervised Dense Event Captioning","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; Canadian Institute for Advanced Research; University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Closed captioning; Computer science; Event (particle physics); Modal; Focus (optics); Feature (linguistics); Representation (politics); Modalities; Ranging; Speech recognition; Audio visual; Natural language processing; Artificial intelligence; Multimedia; Image (mathematics); Linguistics","score_opus":0.023672183288163867,"score_gpt":0.29058380137952466,"score_spread":0.26691161809136077,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2973802306","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.053558446,0.0016999401,0.917724,0.001177705,0.00041263225,0.000405176,0.0037618154,0.014265828,0.006994524],"genre_scores_gemma":[0.54733676,0.0009309738,0.4144725,0.0010878597,0.0008061775,0.0004984767,0.018920653,0.0013232898,0.01462322],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987552,0.00049120886,0.000043348762,0.00042753757,0.0001601857,0.00012250827],"domain_scores_gemma":[0.9977017,0.0011659202,0.00015093302,0.00049948005,0.00031662712,0.00016540398],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016503525,0.0021008644,0.0012447698,0.0012336405,0.000726093,0.0015855053,0.0025816334,0.0022730809,0.0048714955],"category_scores_gemma":[0.005498036,0.0005677054,0.0011260584,0.0013499098,0.0011147403,0.0026950727,0.0024279798,0.002845018,0.002431063],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017425425,0.00046435412,0.001986685,0.0006654278,0.00023095858,0.0006353173,0.00063682016,0.18535179,0.026609343,0.007424237,0.058960225,0.7152923],"study_design_scores_gemma":[0.000040143917,0.000114461494,0.00090467994,0.000034289165,0.00003319106,0.00017122753,0.00013614175,0.9680541,0.010500307,0.012858481,0.0071148444,0.000038043905],"about_ca_topic_score_codex":0.0039777653,"about_ca_topic_score_gemma":0.006096236,"teacher_disagreement_score":0.0048714955,"about_ca_system_score_codex":0.0008778521,"about_ca_system_score_gemma":0.0006948579,"threshold_uncertainty_score":0.016296744},"labels":[],"label_agreement":null},{"id":"W2982151481","doi":"10.1109/iccv.2019.00184","title":"Exploring Overall Contextual Information for Image Captioning in Human-Like Cognitive Style","year":2019,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Closed captioning; Computer science; Sentence; Artificial intelligence; Encoder; ENCODE; Convolutional neural network; Salient; Natural language processing; Cognition; Speech recognition; Image (mathematics)","score_opus":0.04834636649097077,"score_gpt":0.298771803654511,"score_spread":0.2504254371635402,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2982151481","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18823765,0.0047984826,0.7812729,0.0009994151,0.00047418117,0.0004330793,0.0021069525,0.010884285,0.010793071],"genre_scores_gemma":[0.6628211,0.0017834142,0.3240928,0.0006508193,0.0002658548,0.00024902256,0.0041385842,0.00046259194,0.0055358037],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996666,0.00010328047,0.000019342082,0.00012934867,0.000048032995,0.00003347141],"domain_scores_gemma":[0.99926347,0.00032442645,0.000076783384,0.00012795553,0.00016019217,0.000047114874],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006186067,0.0014761406,0.00045052188,0.00055443816,0.00028014526,0.0009775716,0.00087343965,0.0009784204,0.0025712752],"category_scores_gemma":[0.0028212038,0.00027604733,0.0009257433,0.00039408825,0.00046558966,0.0021772897,0.00081955054,0.0011779089,0.0011059754],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006958622,0.00032691495,0.0030650194,0.0012597536,0.00024772092,0.00086868176,0.0012171724,0.087082155,0.1427791,0.006138925,0.019757906,0.7365608],"study_design_scores_gemma":[0.00004159875,0.00042757846,0.0032540632,0.00008853672,0.00019707333,0.00043070602,0.00031576087,0.91142184,0.05773909,0.011014534,0.014994586,0.000074614],"about_ca_topic_score_codex":0.0022697635,"about_ca_topic_score_gemma":0.004192697,"teacher_disagreement_score":0.0025712752,"about_ca_system_score_codex":0.0005335555,"about_ca_system_score_gemma":0.0004709299,"threshold_uncertainty_score":0.008601785},"labels":[],"label_agreement":null},{"id":"W2995439012","doi":"10.1007/978-3-030-58565-5_13","title":"ScanRefer: 3D Object Localization in RGB-D Scans Using Natural Language","year":2020,"lang":"en","type":"preprint","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Object (grammar); Computer science; Artificial intelligence; Point cloud; Minimum bounding box; Task (project management); Computer vision; Natural language; Point (geometry); RGB color model; Bounding overwatch; Pattern recognition (psychology); Sentence; Natural language processing; Image (mathematics); Mathematics; Geometry","score_opus":0.018180355763784246,"score_gpt":0.3058155811202975,"score_spread":0.28763522535651326,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2995439012","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011979168,0.0004168267,0.85314876,0.00019534987,0.00018454177,0.0003556912,0.008800361,0.11934326,0.00557613],"genre_scores_gemma":[0.13015321,0.0007297,0.81691414,0.0004564067,0.00009859832,0.0008928526,0.026498271,0.010718058,0.013538691],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992756,0.00008693347,0.000033952816,0.00021734953,0.00030855706,0.00007757378],"domain_scores_gemma":[0.9994825,0.00013974126,0.000032695418,0.00019249879,0.00011579313,0.00003675649],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004911129,0.0021203223,0.0012793873,0.0022509282,0.00061466085,0.0015525047,0.0023063198,0.0014347667,0.035902016],"category_scores_gemma":[0.0019827404,0.0010251714,0.0014556742,0.0020113518,0.00075853086,0.0022543224,0.0039280495,0.0007850448,0.015760526],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012244761,0.00023358806,0.001759467,0.0012524568,0.00020376885,0.0009414098,0.00067260367,0.010357691,0.14502436,0.008242202,0.09240744,0.7376805],"study_design_scores_gemma":[0.0005565146,0.0007248607,0.00829329,0.00031075507,0.00022028133,0.0028658954,0.0013385334,0.5110391,0.22016758,0.04022578,0.2139008,0.00035662425],"about_ca_topic_score_codex":0.005951275,"about_ca_topic_score_gemma":0.0135099515,"teacher_disagreement_score":0.035902016,"about_ca_system_score_codex":0.00050992,"about_ca_system_score_gemma":0.00090283883,"threshold_uncertainty_score":0.12010425},"labels":[],"label_agreement":null},{"id":"W2996187564","doi":"","title":"Decentralized Distributed PPO: Mastering PointGoal Navigation","year":2020,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Reinforcement learning; Speedup; Leverage (statistics); Robot; Task (project management); Artificial intelligence; Distributed computing; Computation; Autonomous agent; Compass; Real-time computing; Parallel computing; Algorithm","score_opus":0.059373361083821706,"score_gpt":0.1911636216772524,"score_spread":0.13179026059343069,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2996187564","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021514451,0.00017062876,0.9708544,0.00024873097,0.00008197655,0.000059399903,0.00007549031,0.0019049016,0.0050900932],"genre_scores_gemma":[0.78696084,0.00015763976,0.20367168,0.00033287407,0.000067005974,0.0002489961,0.00022053564,0.00032607853,0.008014351],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997292,0.000061746665,0.000010518413,0.00008949735,0.000059722737,0.000049347702],"domain_scores_gemma":[0.99945754,0.00022926183,0.00004889611,0.000109805034,0.00008764721,0.00006687067],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000663964,0.0006651113,0.0008359198,0.00025581612,0.00037871118,0.0006482636,0.0019559928,0.0009460453,0.0039158575],"category_scores_gemma":[0.0019814426,0.00047561916,0.0004549814,0.00029376018,0.00085664104,0.0010359905,0.0015238944,0.0012851268,0.00082948763],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012458707,0.0000632164,0.00061782124,0.000056040713,0.0000316937,0.000058147773,0.00006305612,0.9337899,0.0018917297,0.010313976,0.002823455,0.05016639],"study_design_scores_gemma":[0.0000149569905,0.00001591767,0.000029784018,0.0000021165852,0.0000026908863,0.0000070741958,0.0000033853075,0.9953719,0.00030200282,0.0038490014,0.00039903377,0.0000021653987],"about_ca_topic_score_codex":0.004262303,"about_ca_topic_score_gemma":0.0051422715,"teacher_disagreement_score":0.004262303,"about_ca_system_score_codex":0.0006320387,"about_ca_system_score_gemma":0.001517239,"threshold_uncertainty_score":0.01309979},"labels":[],"label_agreement":null},{"id":"W2997498756","doi":"10.71781/10271","title":"Visual question answering with modules and language modeling","year":2019,"lang":"en","type":"dissertation","venue":"Open MIND","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Compute Canada; Nvidia","keywords":"Question answering; Computer science; Natural language processing; Linguistics; Artificial intelligence; Information retrieval; Philosophy","score_opus":0.0147581786918676,"score_gpt":0.3420031289771181,"score_spread":0.3272449502852505,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2997498756","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0076273982,0.00078272057,0.97657806,0.0007482204,0.00007203952,0.00017893082,0.0005571884,0.0071418225,0.006313715],"genre_scores_gemma":[0.26484004,0.001296729,0.7149731,0.0007031138,0.00013640612,0.0005653222,0.0024465695,0.0010890705,0.013949653],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99828976,0.00067974575,0.000099638724,0.0005009641,0.00030034455,0.0001295122],"domain_scores_gemma":[0.99794275,0.0011533641,0.000116570984,0.0004338612,0.00024852518,0.00010486291],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020666223,0.0010198117,0.0007096985,0.0017203678,0.00055459194,0.0045603304,0.0024283668,0.002032563,0.0132738575],"category_scores_gemma":[0.008393456,0.00071438676,0.0032125516,0.001118857,0.0011355825,0.006196272,0.003307073,0.0018731386,0.003947178],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005010372,0.00024783303,0.0038829066,0.0013202385,0.00038290472,0.0005715856,0.0029780634,0.1039213,0.021777924,0.29038852,0.018318444,0.5557093],"study_design_scores_gemma":[0.00004524008,0.00010715504,0.00093099824,0.0001947379,0.00010828753,0.0004080416,0.00042253538,0.6233263,0.009425167,0.3119248,0.053040132,0.00006670218],"about_ca_topic_score_codex":0.0061881877,"about_ca_topic_score_gemma":0.006460973,"teacher_disagreement_score":0.0132738575,"about_ca_system_score_codex":0.0015376703,"about_ca_system_score_gemma":0.001181084,"threshold_uncertainty_score":0.04440552},"labels":[],"label_agreement":null},{"id":"W2997514790","doi":"10.1609/aaai.v34i07.6783","title":"Deep Generative Probabilistic Graph Neural Networks for Scene Graph Generation","year":2020,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Artificial intelligence; Scene graph; Probabilistic logic; Generative model; Graph; Theoretical computer science; Minimum bounding box; Pattern recognition (psychology); Generative grammar; Image (mathematics)","score_opus":0.09663262041000328,"score_gpt":0.3077191104554348,"score_spread":0.21108649004543154,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2997514790","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012733645,0.00052373804,0.9759931,0.00046224173,0.00008088889,0.000116074574,0.0008456199,0.0070087267,0.0022359649],"genre_scores_gemma":[0.41281572,0.00054243306,0.5704845,0.0010814556,0.0001009989,0.0003766104,0.0067684385,0.0013539406,0.006475891],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999471,0.00010180703,0.000017829178,0.00024649512,0.00010248425,0.00006036154],"domain_scores_gemma":[0.9992747,0.00035885614,0.000065608125,0.00014284096,0.00011331311,0.00004469492],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006086453,0.0017122438,0.0010939093,0.0011041794,0.0005104776,0.00094984676,0.0030927677,0.0018103445,0.0049767415],"category_scores_gemma":[0.0025316097,0.00088345865,0.0019419158,0.0010843694,0.0009806022,0.0025010328,0.0014796152,0.002649728,0.0013322968],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012324304,0.00009843989,0.0011564883,0.00018910438,0.00010916438,0.00019173787,0.00012879328,0.7411143,0.005536309,0.024152834,0.0116096595,0.21558991],"study_design_scores_gemma":[0.000010789213,0.0000136326,0.00008013769,0.0000068340146,0.000009519976,0.000028278157,0.000008450414,0.97838116,0.0010977,0.019153811,0.0012033619,0.000006231301],"about_ca_topic_score_codex":0.012749316,"about_ca_topic_score_gemma":0.021529235,"teacher_disagreement_score":0.012749316,"about_ca_system_score_codex":0.0021592593,"about_ca_system_score_gemma":0.0011551584,"threshold_uncertainty_score":0.025350213},"labels":[],"label_agreement":null},{"id":"W2997948740","doi":"10.1609/aaai.v34i07.6893","title":"Region-Based Global Reasoning Networks","year":2020,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Higher Education Discipline Innovation Project; National Natural Science Foundation of China","keywords":"Computer science; Benchmark (surveying); Focus (optics); Visual reasoning; Artificial intelligence; Pixel; Semantics (computer science); Action (physics); Machine learning; Geography; Cartography","score_opus":0.07076003573598823,"score_gpt":0.3006842134052787,"score_spread":0.22992417766929046,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2997948740","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014586561,0.00055745087,0.97921705,0.00024529395,0.000027757973,0.00006971024,0.00035925614,0.0018352132,0.0031016222],"genre_scores_gemma":[0.50448644,0.0009452432,0.4827731,0.0003818888,0.0000883681,0.00018204747,0.0019168442,0.00035771032,0.008868365],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989309,0.00012905271,0.000054308694,0.0005392976,0.00024852488,0.000097973476],"domain_scores_gemma":[0.9988739,0.0003637986,0.00019302146,0.00024128973,0.00025545206,0.00007252484],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012105792,0.0016148922,0.0010516108,0.001864012,0.0007821881,0.0019304914,0.0034244661,0.001923851,0.00538397],"category_scores_gemma":[0.0034566133,0.0007096525,0.0017127785,0.0015716426,0.0011115745,0.0045995493,0.0020158668,0.0016746281,0.0013737475],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003031903,0.00014607342,0.0026460649,0.0002871808,0.00020000822,0.0003611961,0.00053817703,0.48764178,0.014058017,0.051797222,0.0074859275,0.43453512],"study_design_scores_gemma":[0.000014996359,0.000042357242,0.00052448665,0.000026080908,0.000073389085,0.000085696025,0.000058257487,0.9538874,0.00455783,0.03713843,0.0035715296,0.000019561063],"about_ca_topic_score_codex":0.009171126,"about_ca_topic_score_gemma":0.01197185,"teacher_disagreement_score":0.009171126,"about_ca_system_score_codex":0.0014120437,"about_ca_system_score_gemma":0.0011306249,"threshold_uncertainty_score":0.018235505},"labels":[],"label_agreement":null},{"id":"W2998392846","doi":"","title":"HoME: a Household Multimodal Environment.","year":2018,"lang":"en","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Université de Sherbrooke","funders":"","keywords":"Computer science; Generalization; Human–computer interaction; Robotics; Context (archaeology); Semantics (computer science); Reinforcement learning; Artificial intelligence; Transfer of learning; Multimodal interaction; Robot; Programming language","score_opus":0.011182462263243625,"score_gpt":0.21810827262158627,"score_spread":0.20692581035834265,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2998392846","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26037434,0.0052415333,0.17043656,0.010713019,0.0026561504,0.0013411671,0.034064885,0.04312707,0.4720452],"genre_scores_gemma":[0.6398962,0.0032632705,0.0586349,0.0021845307,0.001163999,0.0011282162,0.012400487,0.002353898,0.2789745],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99983656,0.0000681456,0.000004515424,0.000029443056,0.000026650596,0.000034580913],"domain_scores_gemma":[0.9997141,0.00007362274,0.000009777143,0.000028227498,0.000048938895,0.00012533841],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00029470696,0.0007449225,0.0004729412,0.00029236346,0.0013397512,0.0018164193,0.000718499,0.0012652298,0.1145855],"category_scores_gemma":[0.00064311444,0.00017653276,0.000281087,0.0006135546,0.00043668892,0.0016611859,0.0030908333,0.00054022705,0.015090258],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023141361,0.00034727788,0.004052521,0.0010079412,0.000054798194,0.0052422974,0.0071828067,0.0031550264,0.028710226,0.010380442,0.6407592,0.29679334],"study_design_scores_gemma":[0.00019295044,0.00074734003,0.013352006,0.0005021506,0.00011243189,0.0049295523,0.012023546,0.0138315745,0.009238425,0.006974168,0.93789184,0.0002039666],"about_ca_topic_score_codex":0.002212543,"about_ca_topic_score_gemma":0.0062059015,"teacher_disagreement_score":0.1145855,"about_ca_system_score_codex":0.00034080018,"about_ca_system_score_gemma":0.0003773937,"threshold_uncertainty_score":0.38332665},"labels":[],"label_agreement":null},{"id":"W3000176874","doi":"10.1109/iccv.2019.00999","title":"LayoutVAE: Stochastic Scene Layout Generation From a Label Set","year":2019,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":140,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; Simon Fraser University","funders":"","keywords":"MNIST database; Computer science; Set (abstract data type); Image (mathematics); Artificial intelligence; Autoencoder; Theoretical computer science; Computer vision; Deep learning; Programming language","score_opus":0.02946545474529133,"score_gpt":0.28468178264534616,"score_spread":0.25521632790005483,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3000176874","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013917592,0.00021251028,0.9751902,0.00020857551,0.00010898667,0.00012897514,0.0009796078,0.007933493,0.0013199499],"genre_scores_gemma":[0.22799098,0.0001933483,0.7554896,0.00046111466,0.0001112466,0.00038074353,0.0070820046,0.0018900724,0.006400972],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995067,0.00011539411,0.000014667022,0.0002197165,0.000095449286,0.000048051243],"domain_scores_gemma":[0.9991423,0.0003589318,0.00006617316,0.00019359548,0.00017210185,0.00006688031],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00088213553,0.0013424328,0.00082834944,0.00096107967,0.00045640595,0.00088928605,0.0019495896,0.0015080796,0.005195728],"category_scores_gemma":[0.0027878867,0.0008857479,0.0012924506,0.0006940002,0.0007061141,0.0012516159,0.0014619429,0.0021085618,0.0020356355],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036421465,0.00021212938,0.0021139975,0.00031129035,0.00012885239,0.00026569434,0.00022792631,0.5644703,0.028184846,0.014801519,0.02788558,0.36103365],"study_design_scores_gemma":[0.000026119742,0.00003667172,0.00013463352,0.000009842827,0.0000060667553,0.00003329776,0.000016081964,0.9896354,0.0031660884,0.0051401174,0.0017860376,0.000009576987],"about_ca_topic_score_codex":0.005455276,"about_ca_topic_score_gemma":0.013938149,"teacher_disagreement_score":0.005455276,"about_ca_system_score_codex":0.0011228605,"about_ca_system_score_gemma":0.0010243151,"threshold_uncertainty_score":0.01738149},"labels":[],"label_agreement":null},{"id":"W3003423830","doi":"10.1109/tmm.2020.2971171","title":"Dual Convolutional LSTM Network for Referring Image Segmentation","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Multimedia","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China","keywords":"Encoder; Focus (optics); Dual (grammatical number); Segmentation; Image segmentation; Intersection (aeronautics); Natural language; Object (grammar)","score_opus":0.02954615709882722,"score_gpt":0.2871377927065734,"score_spread":0.25759163560774617,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3003423830","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04225985,0.0023872284,0.9397529,0.00074784487,0.00020806084,0.00008602923,0.0007107301,0.0069843153,0.006863019],"genre_scores_gemma":[0.7037888,0.0014990286,0.2761419,0.0010700953,0.00016903409,0.00016441446,0.0024216953,0.00039061674,0.014354417],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999721,0.00004756208,0.000012771651,0.00011575252,0.000050874478,0.00005206564],"domain_scores_gemma":[0.9998522,0.000045838806,0.000024474519,0.00002461865,0.000040665655,0.000012294782],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00034313067,0.00089209055,0.00054473977,0.00058713654,0.00025142785,0.00061596656,0.0013006856,0.0015021196,0.003378274],"category_scores_gemma":[0.0009974118,0.00029836182,0.00073560723,0.00087238254,0.00045613432,0.0014405862,0.0007029035,0.0009512772,0.0011375265],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005839942,0.0002113753,0.0010958845,0.0004305826,0.00022093834,0.00081910496,0.0003461066,0.25863203,0.11092533,0.014888873,0.019482585,0.59236324],"study_design_scores_gemma":[0.000008179728,0.000052013238,0.00040541438,0.000014851009,0.000045260225,0.00013139241,0.000024838573,0.9763698,0.014497757,0.0054241875,0.003010456,0.000015867361],"about_ca_topic_score_codex":0.0059010917,"about_ca_topic_score_gemma":0.007851456,"teacher_disagreement_score":0.0059010917,"about_ca_system_score_codex":0.00092059723,"about_ca_system_score_gemma":0.0007575347,"threshold_uncertainty_score":0.011733472},"labels":[],"label_agreement":null},{"id":"W3004831296","doi":"10.1109/bibm47256.2019.8983272","title":"Multitask and Multimodal Neural Network Model for Interpretable Analysis of X-ray Images","year":2019,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Optech (Canada)","funders":"","keywords":"Interpretability; Computer science; Artificial intelligence; Closed captioning; Artificial neural network; Pattern recognition (psychology); Natural language processing; State (computer science); Image (mathematics)","score_opus":0.008721184977760824,"score_gpt":0.27077729390256783,"score_spread":0.262056108924807,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3004831296","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22028096,0.001930457,0.7612902,0.002389042,0.0003016099,0.00022229417,0.0031148142,0.003914736,0.0065559247],"genre_scores_gemma":[0.9283709,0.00037398678,0.060111713,0.00032078664,0.0001724042,0.00033437525,0.0014439404,0.00010798918,0.008763828],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997377,0.00008107014,0.000015144041,0.000097411736,0.00003450035,0.00003431004],"domain_scores_gemma":[0.9992811,0.00043645294,0.00008343033,0.00004240666,0.00012875824,0.000027896343],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009390469,0.00092706294,0.0005401964,0.00070832006,0.00028865185,0.00065231585,0.0011368688,0.0012498779,0.0027127762],"category_scores_gemma":[0.0024164577,0.0002859241,0.0008129981,0.0005678783,0.00036144376,0.0008095055,0.00055484974,0.0011590923,0.00078218034],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035236654,0.00025283956,0.0028801053,0.00012003837,0.00011298091,0.00028242063,0.00016172428,0.86558264,0.0054344037,0.0034199748,0.004605174,0.11679545],"study_design_scores_gemma":[0.000002564259,0.000011860181,0.00023431604,0.000003116603,0.0000059608647,0.000009933406,0.000004159029,0.99836236,0.00021073698,0.0010233764,0.00012858429,0.0000030424799],"about_ca_topic_score_codex":0.011035183,"about_ca_topic_score_gemma":0.011303678,"teacher_disagreement_score":0.011035183,"about_ca_system_score_codex":0.0012083356,"about_ca_system_score_gemma":0.0005846822,"threshold_uncertainty_score":0.0219419},"labels":[],"label_agreement":null},{"id":"W3012812419","doi":"10.1109/iros45743.2020.9340914","title":"One-Shot Informed Robotic Visual Search in the Wild","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Vector Institute; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Australian Centre for Field Robotics","keywords":"Computer science; Artificial intelligence; Robot; Visual search; Computer vision; Viewpoints; Task (project management); Field (mathematics); Similarity (geometry); Human–computer interaction; Focus (optics); Flexibility (engineering); Robotics; Image (mathematics)","score_opus":0.11714171087324242,"score_gpt":0.38875510885397835,"score_spread":0.2716133979807359,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3012812419","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21712248,0.0007167037,0.77188766,0.00029846953,0.00009853065,0.0001628142,0.0005318228,0.0042890697,0.0048924545],"genre_scores_gemma":[0.7999731,0.00014697597,0.19462374,0.00020497476,0.00003430474,0.00009198925,0.0010894536,0.00030266176,0.0035327831],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99939156,0.00013465657,0.000022254797,0.00024815887,0.0001354364,0.000067897614],"domain_scores_gemma":[0.9992563,0.00031341065,0.00007107363,0.0002008437,0.000094805466,0.00006349659],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00074813457,0.0008378141,0.0010928202,0.00048569674,0.00040932486,0.00082646904,0.0018927373,0.0014780556,0.0026368105],"category_scores_gemma":[0.0035169655,0.0004939281,0.00054349663,0.00051152235,0.0010150432,0.0024592269,0.001813457,0.00093712576,0.00064979604],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013139428,0.0005386374,0.0020105608,0.00035244247,0.00016870056,0.00052384526,0.0006734683,0.61543643,0.058118414,0.009332519,0.0071564536,0.30437458],"study_design_scores_gemma":[0.00003618467,0.0001518371,0.0005316082,0.000009432532,0.000009962303,0.000116054616,0.000089564404,0.983519,0.0061405976,0.008331476,0.0010439638,0.000020278589],"about_ca_topic_score_codex":0.007013303,"about_ca_topic_score_gemma":0.0070378957,"teacher_disagreement_score":0.007013303,"about_ca_system_score_codex":0.0005918866,"about_ca_system_score_gemma":0.0008218971,"threshold_uncertainty_score":0.013944924},"labels":[],"label_agreement":null},{"id":"W3014594587","doi":"10.1007/978-3-030-58586-0_27","title":"ProxyNCA++: Revisiting and Revitalizing Proxy Neighborhood Component Analysis","year":2020,"lang":"en","type":"preprint","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph; Vector Institute","funders":"","keywords":"Pooling; Proxy (statistics); Scaling; Metric (unit); Computer science; Similarity (geometry); Recall; Precision and recall; Task (project management); Component (thermodynamics); Artificial intelligence; Algorithm; Machine learning; Pattern recognition (psychology); Mathematics; Image (mathematics); Psychology","score_opus":0.01792349795523812,"score_gpt":0.29121482040543406,"score_spread":0.27329132245019594,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3014594587","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0027909833,0.00068382354,0.9883249,0.00028131588,0.00028650017,0.000080070946,0.00022056195,0.005595619,0.0017362818],"genre_scores_gemma":[0.04670907,0.0009645274,0.93968475,0.00039128851,0.00032888076,0.00020537109,0.0011488913,0.004269622,0.006297638],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9967624,0.0010939583,0.00016212332,0.0005547965,0.0012795501,0.00014709613],"domain_scores_gemma":[0.993411,0.0016411286,0.00014102917,0.0028190967,0.0017650946,0.00022265027],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030871625,0.0021234227,0.0026134676,0.0021005285,0.0014418385,0.0047229775,0.0051917764,0.0019003222,0.008717662],"category_scores_gemma":[0.019675074,0.0012676645,0.0019716835,0.0028453113,0.0017692375,0.0040504984,0.006227672,0.0037072897,0.007438551],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033746194,0.00019054058,0.0017539241,0.0004299315,0.00027135544,0.00018242055,0.00028518058,0.060669977,0.0067096483,0.06620942,0.04025554,0.8227046],"study_design_scores_gemma":[0.00004946178,0.000063638116,0.0003815044,0.000067202614,0.00008306047,0.00013567897,0.00006385934,0.90422225,0.0048720753,0.03707827,0.05292455,0.000058491863],"about_ca_topic_score_codex":0.018090812,"about_ca_topic_score_gemma":0.026231572,"teacher_disagreement_score":0.018090812,"about_ca_system_score_codex":0.0010449622,"about_ca_system_score_gemma":0.0034338976,"threshold_uncertainty_score":0.035971045},"labels":[],"label_agreement":null},{"id":"W3014867697","doi":"10.3390/sym12040511","title":"Attentive Gated Graph Neural Network for Image Scene Graph Generation","year":2020,"lang":"en","type":"article","venue":"Symmetry","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"National Natural Science Foundation of China","keywords":"Computer science; Artificial intelligence; Pattern recognition (psychology); Embedding; Graph; Artificial neural network; Scene graph; Visualization; Image (mathematics); Relation (database); Graph embedding; Computer vision; Theoretical computer science; Data mining","score_opus":0.026025546355178602,"score_gpt":0.27781754890669424,"score_spread":0.2517920025515156,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3014867697","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032551654,0.00044385568,0.9587134,0.00036038665,0.00009577135,0.00014713642,0.00054439285,0.003999783,0.0031436796],"genre_scores_gemma":[0.641369,0.0005085768,0.3451453,0.0006989845,0.000081257596,0.00026097245,0.0032663366,0.00045285813,0.0082167825],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998635,0.000024696837,0.0000045113275,0.000055497603,0.00002899339,0.0000227316],"domain_scores_gemma":[0.9998254,0.000059021088,0.000019911993,0.000030980347,0.000046933765,0.000017750115],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00021635502,0.00096564135,0.000535813,0.0006894535,0.0002444441,0.00036332518,0.0014725737,0.0008152719,0.003832724],"category_scores_gemma":[0.0008408882,0.00033794026,0.00062824547,0.00072176155,0.0004424179,0.000994269,0.0007064086,0.0011117053,0.00074502046],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019193822,0.00021417157,0.00084521505,0.00014722801,0.00006725628,0.00023987245,0.00009194842,0.53324497,0.027554985,0.018063128,0.01358879,0.40575054],"study_design_scores_gemma":[0.000006215245,0.000018722632,0.00011018498,0.000002990791,0.000006365251,0.000017570614,0.0000057190446,0.9922161,0.0017566779,0.005283785,0.00057160814,0.0000039367233],"about_ca_topic_score_codex":0.008622589,"about_ca_topic_score_gemma":0.014717512,"teacher_disagreement_score":0.008622589,"about_ca_system_score_codex":0.00088138354,"about_ca_system_score_gemma":0.0007159284,"threshold_uncertainty_score":0.0171448},"labels":[],"label_agreement":null},{"id":"W3014973596","doi":"10.48550/arxiv.2004.00760","title":"Consistent Multiple Sequence Decoding","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Decoding methods; Computer science; Fusion mechanism; Context (archaeology); Sequential decoding; List decoding; Sequence (biology); Closed captioning; Task (project management); Algorithm; Pattern recognition (psychology); Artificial intelligence; Theoretical computer science; Fusion; Image (mathematics); Concatenated error correction code; Block code","score_opus":0.1794902213038457,"score_gpt":0.22722523727327126,"score_spread":0.04773501596942556,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3014973596","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006859002,0.00024279964,0.984993,0.0002872898,0.00011896588,0.00006660207,0.0002803009,0.0029700838,0.004181978],"genre_scores_gemma":[0.36653578,0.0004369816,0.6157286,0.0005865743,0.00018093454,0.00024415483,0.0018719439,0.0016663838,0.012748656],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99819416,0.0005017528,0.00011853689,0.0005467821,0.00049938646,0.00013939393],"domain_scores_gemma":[0.9963003,0.0012075622,0.00022790686,0.0011854301,0.00096169394,0.00011721988],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016726474,0.0012658264,0.0012300372,0.0009180309,0.00076688407,0.0019182449,0.0028834874,0.0017974875,0.008007876],"category_scores_gemma":[0.008788085,0.00052098883,0.0011305279,0.0012058418,0.0015840665,0.003793302,0.0026653353,0.0021904756,0.0044506337],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005315589,0.00016643373,0.0018368826,0.00045112945,0.00015474409,0.00056042033,0.0005663344,0.18839943,0.03354048,0.16010621,0.019505542,0.5941809],"study_design_scores_gemma":[0.000025526977,0.00010449893,0.00022161868,0.000039708135,0.00003655859,0.00028688315,0.00007810324,0.8414507,0.028468398,0.120223045,0.009023246,0.000041737014],"about_ca_topic_score_codex":0.0040760334,"about_ca_topic_score_gemma":0.0061957343,"teacher_disagreement_score":0.008007876,"about_ca_system_score_codex":0.0010051017,"about_ca_system_score_gemma":0.0024737394,"threshold_uncertainty_score":0.02678907},"labels":[],"label_agreement":null},{"id":"W3015882298","doi":"10.36227/techrxiv.12093564.v1","title":"Image Captioning with Complementary Visual and Textual Cues","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Lakehead University","funders":"","keywords":"Closed captioning; Modality (human–computer interaction); Image (mathematics); Computer science; Artificial intelligence; Computer vision; Natural language processing","score_opus":0.021762578800096587,"score_gpt":0.31395479874815174,"score_spread":0.29219221994805517,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3015882298","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03141645,0.0027332471,0.91609484,0.0015702153,0.0021444869,0.00048704233,0.0026290342,0.0114792185,0.03144547],"genre_scores_gemma":[0.34790894,0.0030500328,0.6019964,0.0015208351,0.0020830454,0.0006875116,0.007673557,0.0026215182,0.032458115],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9992034,0.00023723688,0.000051126764,0.00020044502,0.00022178187,0.0000860953],"domain_scores_gemma":[0.9977012,0.0007729089,0.00013085899,0.0005237695,0.00073467457,0.00013664007],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008312141,0.0016193789,0.0008993992,0.0013021537,0.0005587499,0.0022649136,0.0011042026,0.0020077475,0.02615845],"category_scores_gemma":[0.005603916,0.0004683754,0.0011255115,0.0013000427,0.00065894105,0.003096398,0.0022222744,0.0017532945,0.00763208],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015500129,0.00025197212,0.00031939498,0.0017389518,0.0001298352,0.0011255901,0.00035957873,0.021545462,0.25367063,0.010145276,0.07230926,0.636854],"study_design_scores_gemma":[0.00014422902,0.0006311253,0.0015781422,0.00031217444,0.00027505012,0.0020491995,0.00048020258,0.5442512,0.34653106,0.022127457,0.08142152,0.00019871326],"about_ca_topic_score_codex":0.0007611735,"about_ca_topic_score_gemma":0.00094222586,"teacher_disagreement_score":0.02615845,"about_ca_system_score_codex":0.00042731807,"about_ca_system_score_gemma":0.00039605497,"threshold_uncertainty_score":0.08750874},"labels":[],"label_agreement":null},{"id":"W3031234741","doi":"10.1145/3318464.3384701","title":"SVQ++: Querying for Object Interactions in Video Streams","year":2020,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Frame (networking); Object (grammar); Video tracking; Artificial intelligence; Computer vision; Video processing; Object detection; Frame rate; Pattern recognition (psychology)","score_opus":0.03883423857464274,"score_gpt":0.32772350611831785,"score_spread":0.2888892675436751,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3031234741","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05100855,0.00086058985,0.84680957,0.00047517815,0.00017685922,0.00040156752,0.007968755,0.08683278,0.0054661403],"genre_scores_gemma":[0.3396341,0.00067585235,0.63331515,0.00048748488,0.00011780742,0.00047063731,0.015385829,0.0022170704,0.0076960223],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9992907,0.00009433541,0.000044587676,0.00020468787,0.0003004728,0.00006529166],"domain_scores_gemma":[0.99927276,0.0003538144,0.000057558016,0.00012573865,0.00013295424,0.00005723517],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000991376,0.001237424,0.0010321072,0.001096984,0.00047664266,0.0016777122,0.0027646285,0.0010878972,0.0070897825],"category_scores_gemma":[0.0030664094,0.00059724256,0.00057217234,0.0013771744,0.0006100258,0.0026265325,0.002055786,0.0011251608,0.0017426793],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025083884,0.00036350684,0.006517188,0.0007628726,0.00032530795,0.0007221927,0.000740143,0.055768438,0.08009457,0.022121863,0.10708924,0.72298634],"study_design_scores_gemma":[0.0001095572,0.00013742954,0.0015354382,0.000021730477,0.000030354598,0.00020086301,0.00011069365,0.93509644,0.0320041,0.010727886,0.019974964,0.000050486815],"about_ca_topic_score_codex":0.012908291,"about_ca_topic_score_gemma":0.018081399,"teacher_disagreement_score":0.012908291,"about_ca_system_score_codex":0.00097621407,"about_ca_system_score_gemma":0.00086010405,"threshold_uncertainty_score":0.025666296},"labels":[],"label_agreement":null},{"id":"W3034585290","doi":"10.1109/cvpr42600.2020.00365","title":"Composed Query Image Retrieval Using Locally Bounded Features","year":2020,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":80,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Computer science; Image retrieval; Image (mathematics); Margin (machine learning); Benchmark (surveying); Set (abstract data type); Visual Word; Artificial intelligence; Information retrieval; Pattern recognition (psychology); Sentence; Machine learning","score_opus":0.025373388347400473,"score_gpt":0.28916207059125776,"score_spread":0.26378868224385726,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3034585290","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08886695,0.0041450965,0.89253026,0.00046827932,0.00016490405,0.00039784805,0.0014691756,0.007143081,0.0048144786],"genre_scores_gemma":[0.64574265,0.0013693105,0.3328021,0.00062538736,0.00041122668,0.00039916075,0.0062514916,0.0005911581,0.011807483],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989875,0.00014860812,0.00006560772,0.00034788085,0.00034352718,0.00010682884],"domain_scores_gemma":[0.99911755,0.00023978668,0.00011211364,0.00027906636,0.0002033782,0.000048102396],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006596395,0.0013280615,0.0020332711,0.002145009,0.00044636917,0.001210235,0.0017612762,0.0013122782,0.0046017505],"category_scores_gemma":[0.00242358,0.0003109969,0.001045616,0.0022998794,0.0006045768,0.0031376518,0.0014078108,0.0007945354,0.002155738],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017246716,0.0005044505,0.0020938958,0.00071127777,0.00031124035,0.00096734375,0.00029994056,0.06352644,0.15189876,0.0087845735,0.03140452,0.73777294],"study_design_scores_gemma":[0.000117918986,0.00040912593,0.002325412,0.000028635164,0.00016196382,0.0008958992,0.00017397224,0.94333583,0.032136284,0.011608218,0.008741535,0.00006527892],"about_ca_topic_score_codex":0.0051908945,"about_ca_topic_score_gemma":0.0062119863,"teacher_disagreement_score":0.0051908945,"about_ca_system_score_codex":0.0009170967,"about_ca_system_score_gemma":0.00071148964,"threshold_uncertainty_score":0.01539439},"labels":[],"label_agreement":null},{"id":"W3034733309","doi":"10.18653/v1/2020.acl-main.664","title":"Improving Image Captioning with Better Use of Caption","year":2020,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":85,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Closed captioning; Computer science; Leverage (statistics); Artificial intelligence; Natural language processing; Semantics (computer science); Visualization; Feature learning; Representation (politics); Word (group theory); Inductive bias; Image (mathematics); Machine learning; Task (project management); Multi-task learning","score_opus":0.02151848015296971,"score_gpt":0.23068016380181638,"score_spread":0.20916168364884669,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3034733309","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019858588,0.0035705864,0.92722416,0.0014438902,0.0010561036,0.0006021062,0.002563799,0.031604756,0.012075931],"genre_scores_gemma":[0.20816353,0.0020059552,0.76129204,0.0014329926,0.0006837877,0.00047342933,0.012600623,0.003289548,0.010058075],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99856585,0.0004609476,0.00006705638,0.00050994696,0.00028962572,0.00010662368],"domain_scores_gemma":[0.9966292,0.0013121428,0.00026532682,0.000988109,0.000664509,0.00014080963],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018852395,0.0026611912,0.0012867844,0.002232692,0.00082495564,0.0029134012,0.0028262376,0.0027001488,0.007712781],"category_scores_gemma":[0.00989101,0.0006015322,0.0016638214,0.0020693757,0.0011612551,0.0049794903,0.0028131946,0.0035967282,0.0061079073],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043079944,0.00035425852,0.0012002066,0.0009875093,0.00019142935,0.00043585847,0.00046237677,0.09661594,0.04339142,0.0121229105,0.10020769,0.7435996],"study_design_scores_gemma":[0.00007124173,0.00017264365,0.00057413825,0.00011044852,0.00011263444,0.00047244152,0.00015347177,0.89223474,0.046863176,0.02222504,0.03694197,0.00006809171],"about_ca_topic_score_codex":0.0026692813,"about_ca_topic_score_gemma":0.0034919875,"teacher_disagreement_score":0.007712781,"about_ca_system_score_codex":0.0012188276,"about_ca_system_score_gemma":0.00089544116,"threshold_uncertainty_score":0.025801837},"labels":[],"label_agreement":null},{"id":"W3035931148","doi":"10.48550/arxiv.2006.10923","title":"Hyperparameter Analysis for Image Captioning","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Closed captioning; Hyperparameter; Computer science; Transformer; Baseline (sea); Encoder; Artificial intelligence; Image (mathematics); Sensitivity (control systems); Preprocessor; Machine learning; Speech recognition; Pattern recognition (psychology); Natural language processing; Engineering","score_opus":0.09702672678732742,"score_gpt":0.2241226351722768,"score_spread":0.12709590838494939,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3035931148","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13305631,0.011190822,0.82600445,0.0019616794,0.0005963835,0.0005309307,0.0021381832,0.008462936,0.016058251],"genre_scores_gemma":[0.90158534,0.00215223,0.08596893,0.00094275427,0.0002877818,0.00037329062,0.0030914652,0.0018733409,0.0037248584],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9954934,0.0021272297,0.00022166039,0.00076682284,0.001124591,0.00026625642],"domain_scores_gemma":[0.9829623,0.012446407,0.0006852963,0.0023811925,0.0013914603,0.00013340761],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005320746,0.001911039,0.0009325316,0.0020190226,0.0005985969,0.0019854854,0.00137271,0.0019765631,0.006455848],"category_scores_gemma":[0.04991529,0.00063179113,0.0012833055,0.0012874268,0.0010188795,0.003053768,0.0019120378,0.002690295,0.0017155969],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00074292853,0.00017126062,0.00421969,0.0011434369,0.0005839842,0.00045347423,0.00025675033,0.63586676,0.048378095,0.0099442005,0.011467828,0.28677163],"study_design_scores_gemma":[0.000021991244,0.00017731833,0.0035333915,0.00013476436,0.00011700839,0.0005124919,0.000085655396,0.94502085,0.027568305,0.016397577,0.00635834,0.00007230132],"about_ca_topic_score_codex":0.0028390249,"about_ca_topic_score_gemma":0.0019435226,"teacher_disagreement_score":0.006455848,"about_ca_system_score_codex":0.0018455952,"about_ca_system_score_gemma":0.0005379044,"threshold_uncertainty_score":0.028139114},"labels":[],"label_agreement":null},{"id":"W3037533539","doi":"10.1609/aaai.v34i07.6833","title":"Learning Cross-Modal Context Graph for Visual Grounding","year":2020,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":84,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China; National Science Foundation","keywords":"Computer science; Ground; Artificial intelligence; Phrase; Graph; Modal; Modular design; Natural language processing; Theoretical computer science; Programming language","score_opus":0.08715630185034338,"score_gpt":0.3538858751306011,"score_spread":0.2667295732802577,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3037533539","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.057409458,0.00097576994,0.9243962,0.00035908332,0.00012073379,0.00015054997,0.0012774111,0.010926401,0.0043844306],"genre_scores_gemma":[0.6647895,0.00044219592,0.31933233,0.00064305327,0.00010774925,0.00020569922,0.0071598315,0.001036424,0.0062832204],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99924576,0.00010588696,0.000021444155,0.00040664375,0.00011760815,0.00010271925],"domain_scores_gemma":[0.9994197,0.00017625003,0.00007251839,0.0001638936,0.00011600978,0.000051628493],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005870227,0.0018319294,0.001240217,0.002606648,0.0008389349,0.0010237903,0.0029661888,0.0024289098,0.005064636],"category_scores_gemma":[0.0025679215,0.00059732056,0.0014277276,0.0024274543,0.0011095728,0.0033294098,0.002529755,0.0020491583,0.00200907],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038433346,0.0003597562,0.003393093,0.00040066973,0.00021457291,0.0004194688,0.0003613058,0.19474033,0.037667513,0.023942208,0.023647076,0.71446973],"study_design_scores_gemma":[0.000036166395,0.00008797261,0.0008813867,0.00003313951,0.000041722175,0.000117105956,0.000113807015,0.95200586,0.0060086357,0.03704016,0.0036091753,0.000024840545],"about_ca_topic_score_codex":0.010863184,"about_ca_topic_score_gemma":0.022731025,"teacher_disagreement_score":0.010863184,"about_ca_system_score_codex":0.0010964298,"about_ca_system_score_gemma":0.0010014137,"threshold_uncertainty_score":0.021599889},"labels":[],"label_agreement":null},{"id":"W3044731091","doi":"10.48550/arxiv.1909.07459","title":"Bridging Visual Perception with Contextual Semantics for Understanding Robot Manipulation Tasks","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Bridging (networking); Robot; Human–computer interaction; Semantics (computer science); Perception; Tuple; Artificial intelligence; Ontology; Natural language processing; Programming language","score_opus":0.12548312163179495,"score_gpt":0.2426436539427157,"score_spread":0.11716053231092075,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3044731091","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0133163035,0.00029563042,0.98067,0.00039004482,0.000023978911,0.00008036577,0.00031358306,0.0011423074,0.003767782],"genre_scores_gemma":[0.4289432,0.00073730637,0.5672959,0.00016386858,0.000029030476,0.00016528822,0.0010822167,0.0002601022,0.001323066],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99966526,0.00010575074,0.000021006299,0.0001054285,0.00007481942,0.00002779029],"domain_scores_gemma":[0.99945503,0.00021584745,0.000069616464,0.00013886925,0.00008171943,0.000038954066],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00048400846,0.00062444113,0.00026160103,0.0013673889,0.00043360709,0.0015289357,0.00083859166,0.0007588403,0.0027503949],"category_scores_gemma":[0.002414199,0.00032548857,0.00084077084,0.00088713993,0.0013826329,0.0030268217,0.0017632222,0.00089673937,0.00050253764],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020390679,0.00025601604,0.0032943145,0.0008106935,0.00012754029,0.0011334579,0.0046236175,0.097185284,0.05161731,0.3938598,0.0068773697,0.44001073],"study_design_scores_gemma":[0.000028063905,0.00011283295,0.0036970598,0.0002458667,0.00012979626,0.00053847046,0.0021474806,0.4449302,0.024121782,0.47349387,0.050462972,0.000091522554],"about_ca_topic_score_codex":0.0056677177,"about_ca_topic_score_gemma":0.00677842,"teacher_disagreement_score":0.0056677177,"about_ca_system_score_codex":0.0008517377,"about_ca_system_score_gemma":0.00086406217,"threshold_uncertainty_score":0.01126945},"labels":[],"label_agreement":null},{"id":"W3080318437","doi":"10.18653/v1/2021.mrl-1.13","title":"VisualSem: a high-quality knowledge graph for vision and language","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Samsung; European Commission; York University; Samsung Advanced Institute of Technology; Nvidia; Universiteit van Amsterdam; National Science Foundation","keywords":"Computer science; Pipeline (software); Modal; Artificial intelligence; Knowledge graph; Artificial neural network; Quality (philosophy); Natural language processing; Bridging (networking); Graph; Natural language; Information retrieval; Programming language; Theoretical computer science","score_opus":0.024389899715241477,"score_gpt":0.397022387454462,"score_spread":0.37263248773922053,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3080318437","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010307994,0.0011129963,0.6659243,0.0016332198,0.00044494786,0.00047948645,0.12696706,0.16965058,0.023479426],"genre_scores_gemma":[0.08697681,0.0012211921,0.544814,0.00082870084,0.000085900276,0.0007213087,0.33354777,0.01606378,0.015740613],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9993906,0.00011037447,0.000041244755,0.0002034621,0.00021100132,0.000043337473],"domain_scores_gemma":[0.99844104,0.0005101529,0.00007634096,0.00051403244,0.000369706,0.00008879859],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007132816,0.00144284,0.00063984847,0.0029266833,0.00056701084,0.0019002516,0.0022720883,0.0016828313,0.030538343],"category_scores_gemma":[0.0069061005,0.00084012095,0.0012748939,0.0022779861,0.0005736333,0.0040652156,0.0025057634,0.0018399311,0.01518914],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031612907,0.00023072331,0.0015203217,0.0016828877,0.00015923755,0.00041625954,0.00034233913,0.046717446,0.009190937,0.044038244,0.56052303,0.33486247],"study_design_scores_gemma":[0.00017510318,0.00013588146,0.0016528148,0.00029592428,0.00008508761,0.00052104495,0.00021179598,0.33928922,0.013922408,0.13456658,0.5090218,0.00012247043],"about_ca_topic_score_codex":0.01486228,"about_ca_topic_score_gemma":0.033963736,"teacher_disagreement_score":0.030538343,"about_ca_system_score_codex":0.0011304202,"about_ca_system_score_gemma":0.0016314082,"threshold_uncertainty_score":0.10216093},"labels":[],"label_agreement":null},{"id":"W3081752373","doi":"10.5244/c.34.107","title":"Sentence Guided Temporal Modulation for Dynamic Video Thumbnail Generation","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Thumbnail; Computer science; Sentence; Embedding; Modulation (music); Artificial intelligence; Natural language processing; Speech recognition; Image (mathematics)","score_opus":0.06512363023925581,"score_gpt":0.3391107237255148,"score_spread":0.273987093486259,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3081752373","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.060705803,0.0012658718,0.9312602,0.00042775052,0.00020785652,0.00014870569,0.0003794784,0.0028134575,0.002790912],"genre_scores_gemma":[0.724317,0.0004888218,0.27056888,0.00030903963,0.0001617993,0.00015771696,0.00065626967,0.00020006888,0.003140337],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997136,0.0000823867,0.000018379627,0.00010003648,0.0000557744,0.000029870518],"domain_scores_gemma":[0.9995604,0.00021517486,0.000051871833,0.000066979745,0.00007309536,0.000032497734],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00059011084,0.0006167364,0.00045636835,0.00022701696,0.00016572814,0.00034472186,0.00079978973,0.00061204546,0.002134835],"category_scores_gemma":[0.002007412,0.00015912387,0.00038730088,0.0002647157,0.00030103736,0.00074591633,0.00045124182,0.0007786619,0.00051833136],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00053358043,0.00021051544,0.0009958874,0.00039679307,0.0000867345,0.00053127017,0.00027783218,0.109567635,0.21710962,0.010531869,0.009059318,0.6506991],"study_design_scores_gemma":[0.000026683609,0.000200909,0.00045478894,0.000011838318,0.00003054495,0.00019405306,0.000032648357,0.9434458,0.04557334,0.0069996454,0.0030153343,0.000014409161],"about_ca_topic_score_codex":0.0011322119,"about_ca_topic_score_gemma":0.0017290934,"teacher_disagreement_score":0.002134835,"about_ca_system_score_codex":0.0003659802,"about_ca_system_score_gemma":0.00035993045,"threshold_uncertainty_score":0.0071417093},"labels":[],"label_agreement":null},{"id":"W3083845092","doi":"10.48550/arxiv.2009.04806","title":"SketchEmbedNet: Learning Novel Concepts by Imitating Drawings","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Sketch; Computer science; Principle of compositionality; Salient; Generative grammar; Artificial intelligence; Focus (optics); Class (philosophy); Natural (archaeology); Natural language processing; Algorithm","score_opus":0.06690291441134907,"score_gpt":0.22608206366341768,"score_spread":0.1591791492520686,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3083845092","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05114506,0.0012303801,0.9212096,0.00047063426,0.00038422566,0.00031034506,0.0029976682,0.013871324,0.008380767],"genre_scores_gemma":[0.36255586,0.0011891949,0.60973567,0.00044951553,0.0001269042,0.0005489976,0.009407943,0.0011408837,0.014845025],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996731,0.000057132318,0.000011058096,0.00016806334,0.0000679402,0.000022728549],"domain_scores_gemma":[0.9995339,0.000179808,0.00003718575,0.00016396675,0.00004544225,0.000039724175],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005373494,0.001656746,0.0006730722,0.0009868963,0.0002462871,0.0011741198,0.002344389,0.0014583583,0.010765249],"category_scores_gemma":[0.0030574666,0.0006442045,0.0009779106,0.0008770324,0.0006907169,0.0031105117,0.0014469519,0.0016093579,0.0024428975],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036860837,0.00027115364,0.0022261424,0.000743343,0.00026102338,0.0004659507,0.00025389242,0.20191963,0.023076616,0.030810254,0.04543312,0.69417024],"study_design_scores_gemma":[0.000046725454,0.00010082473,0.000376697,0.000038458304,0.000027888302,0.00023301205,0.000048628182,0.95646214,0.007808426,0.02205977,0.012776656,0.000020803263],"about_ca_topic_score_codex":0.0016687687,"about_ca_topic_score_gemma":0.0047097937,"teacher_disagreement_score":0.010765249,"about_ca_system_score_codex":0.00071523886,"about_ca_system_score_gemma":0.00043722478,"threshold_uncertainty_score":0.036013365},"labels":[],"label_agreement":null},{"id":"W3088099248","doi":"","title":"Memory Augmented Neural Networks for Natural Language Processing","year":2017,"lang":"en","type":"article","venue":"Empirical Methods in Natural Language Processing","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Artificial intelligence; Artificial neural network; Scalability; Set (abstract data type); Focus (optics); Auxiliary memory; Deep learning; Machine learning; Theoretical computer science; Programming language","score_opus":0.040286962499015484,"score_gpt":0.46035881602918827,"score_spread":0.4200718535301728,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3088099248","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0058908877,0.028746337,0.9328775,0.003434018,0.0009876898,0.00007014274,0.0007113497,0.002375619,0.024906466],"genre_scores_gemma":[0.35584688,0.04099201,0.53757054,0.0017247502,0.0018400659,0.0006925424,0.002768675,0.00062646484,0.05793811],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997805,0.000055740056,0.000015902588,0.00006147766,0.00006569506,0.000020747295],"domain_scores_gemma":[0.99972004,0.00015564934,0.000022743734,0.000039835417,0.000051435163,0.000010224222],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00038816713,0.00068026705,0.0005380233,0.00045555792,0.00026003493,0.0014295406,0.0011779469,0.001329491,0.008225783],"category_scores_gemma":[0.0016813603,0.00028950375,0.0005790612,0.0008275581,0.00075514684,0.002462023,0.0008592088,0.0025157458,0.001992282],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010642108,0.00004813491,0.00034506354,0.00084133795,0.00012422523,0.00017906893,0.0001333182,0.124098554,0.0056834416,0.46993023,0.031252764,0.36725748],"study_design_scores_gemma":[0.000010655918,0.000033695756,0.00022749769,0.0001019693,0.00002183604,0.00008919403,0.000025122557,0.5015792,0.0017641161,0.4447472,0.051375736,0.000023639384],"about_ca_topic_score_codex":0.0028723248,"about_ca_topic_score_gemma":0.0035836163,"teacher_disagreement_score":0.008225783,"about_ca_system_score_codex":0.0009191722,"about_ca_system_score_gemma":0.00058626704,"threshold_uncertainty_score":0.027517915},"labels":[],"label_agreement":null},{"id":"W3089338039","doi":"10.1093/mnras/stab424","title":"Pix2Prof: fast extraction of sequential information from galaxy imagery via a deep natural language ‘captioning’ model","year":2021,"lang":"en","type":"article","venue":"Monthly Notices of the Royal Astronomical Society","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada; Queen's University; Royal Society; Government of Ontario","keywords":"Galaxy; Deep learning; sort; Surface brightness; Sky; Image (mathematics); Artificial neural network; Leverage (statistics)","score_opus":0.005327329558957749,"score_gpt":0.22736021647365892,"score_spread":0.22203288691470116,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3089338039","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02021985,0.00026580796,0.95041424,0.00039205834,0.000081938466,0.00008620144,0.0021296672,0.024203684,0.002206582],"genre_scores_gemma":[0.26834926,0.00030626546,0.7131332,0.00054686336,0.0000935879,0.0002886029,0.010213591,0.0015299217,0.005538624],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997609,0.000036312606,0.000010333904,0.00009263062,0.00007115918,0.000028632181],"domain_scores_gemma":[0.99955195,0.00014685762,0.000049715632,0.000107517335,0.00010810638,0.000035851954],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00043608242,0.0009108099,0.00050800154,0.0009323028,0.00035328194,0.0010699817,0.0018864988,0.0010204663,0.005978638],"category_scores_gemma":[0.0019376812,0.000593094,0.00089865574,0.0007157913,0.0005370644,0.0019122853,0.0011816353,0.0012691983,0.0027128302],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004863751,0.00019722928,0.0034383705,0.0003409806,0.00012876926,0.00046217596,0.00019905846,0.28161913,0.04339413,0.020658668,0.064893074,0.584182],"study_design_scores_gemma":[0.000007696253,0.000017737302,0.00025696334,0.000007173978,0.00000515217,0.000042183678,0.0000085802,0.9852297,0.005632628,0.005279673,0.0035031014,0.000009463214],"about_ca_topic_score_codex":0.005960196,"about_ca_topic_score_gemma":0.0077310815,"teacher_disagreement_score":0.005978638,"about_ca_system_score_codex":0.0007831525,"about_ca_system_score_gemma":0.0008802963,"threshold_uncertainty_score":0.020000577},"labels":[],"label_agreement":null},{"id":"W3090136599","doi":"10.1145/3377816.3381723","title":"Eye of the mind","year":2020,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Coding (social sciences); World Wide Web; Software; Multimedia; Information retrieval; Human–computer interaction; Data science; Programming language","score_opus":0.01688305020625226,"score_gpt":0.2609230488219535,"score_spread":0.24403999861570125,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3090136599","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022030795,0.017016884,0.06056072,0.10847537,0.004391441,0.00009318338,0.0007339026,0.0018187232,0.78487897],"genre_scores_gemma":[0.58031046,0.01036339,0.030611655,0.046275258,0.00206067,0.00015530088,0.000693468,0.0012712582,0.3282585],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988034,0.00035652932,0.00003112779,0.00036154027,0.00033710734,0.00011042972],"domain_scores_gemma":[0.99803895,0.0006833932,0.00013952426,0.0004118848,0.0004886325,0.00023760578],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001012289,0.0005844396,0.00035031483,0.0010112893,0.0024109099,0.0066567496,0.00076125545,0.0031667994,0.041167136],"category_scores_gemma":[0.009170173,0.00032678826,0.0004936671,0.00044762634,0.0060460516,0.0062775156,0.0033698156,0.0034867364,0.014680055],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021107947,0.000053966425,0.0048534595,0.00051452935,0.0000983239,0.00084184384,0.0281585,0.00065768365,0.010918361,0.37001947,0.3555469,0.22812596],"study_design_scores_gemma":[0.00002440909,0.000041907017,0.002279283,0.00043150742,0.000027831109,0.0011214637,0.006747544,0.0008825948,0.0012188121,0.14317095,0.84400356,0.00005013275],"about_ca_topic_score_codex":0.005486388,"about_ca_topic_score_gemma":0.004386529,"teacher_disagreement_score":0.041167136,"about_ca_system_score_codex":0.0014838976,"about_ca_system_score_gemma":0.0016421112,"threshold_uncertainty_score":0.13771778},"labels":[],"label_agreement":null},{"id":"W3092352310","doi":"10.1109/case48305.2020.9216770","title":"Bridging Visual Perception with Contextual Semantics for Understanding Robot Manipulation Tasks","year":2020,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Bridging (networking); Robot; Semantics (computer science); Human–computer interaction; Tuple; Perception; Artificial intelligence; Ontology; Natural language processing; Programming language","score_opus":0.0795439621350606,"score_gpt":0.31188448238756744,"score_spread":0.23234052025250684,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3092352310","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013740446,0.00028903596,0.9802663,0.00035899362,0.000024099507,0.00009014113,0.00029310727,0.0011598064,0.0037781636],"genre_scores_gemma":[0.42852017,0.0007250016,0.56789666,0.0001601408,0.000026473743,0.00016532099,0.0009906993,0.0002535467,0.0012619484],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99967146,0.00010141092,0.00002186474,0.000102084436,0.00007448492,0.000028764018],"domain_scores_gemma":[0.9994667,0.00020694402,0.00007077019,0.00013239215,0.000082633545,0.000040482817],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00048052185,0.0006277924,0.00026057247,0.0013912289,0.00044596937,0.00151338,0.00082354504,0.0006980012,0.0026136504],"category_scores_gemma":[0.0023444346,0.00032120035,0.0008746385,0.00081482227,0.0013324985,0.0030119366,0.0016950817,0.00086385506,0.00045056827],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021260239,0.00027567998,0.0034212563,0.0008981318,0.00013202896,0.0013453778,0.005494703,0.09835113,0.058764882,0.3881592,0.006353854,0.43659124],"study_design_scores_gemma":[0.000032500393,0.00014990098,0.004229648,0.00032095128,0.00016217036,0.0006912712,0.0029105647,0.4460446,0.029542036,0.45685583,0.058946762,0.00011379724],"about_ca_topic_score_codex":0.0059977765,"about_ca_topic_score_gemma":0.0076723625,"teacher_disagreement_score":0.0059977765,"about_ca_system_score_codex":0.0008225753,"about_ca_system_score_gemma":0.0009374388,"threshold_uncertainty_score":0.011925757},"labels":[],"label_agreement":null},{"id":"W3095478416","doi":"10.1109/cvprw53098.2021.00181","title":"An Improved Attention for Visual Question Answering","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; Canadian Institute for Advanced Research; University of British Columbia","funders":"","keywords":"Question answering; Computer science; Benchmark (surveying); Artificial intelligence; Encoder; Context (archaeology); Task (project management); Natural language; Modality (human–computer interaction); Relation (database); Information retrieval; Natural language processing; Data mining","score_opus":0.014382356879745518,"score_gpt":0.349374374597222,"score_spread":0.33499201771747644,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3095478416","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024027579,0.003949417,0.95413077,0.0011694417,0.00029305104,0.00027213278,0.00097591773,0.009835903,0.0053457883],"genre_scores_gemma":[0.52065647,0.0017342416,0.45477784,0.002323124,0.0006715441,0.00040173068,0.005297131,0.0005715585,0.013566396],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989318,0.00023559638,0.000048671598,0.00044627613,0.00019075247,0.00014685094],"domain_scores_gemma":[0.99929523,0.00029318134,0.000032853655,0.00011669991,0.00020576916,0.000056258777],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010856992,0.0013423753,0.0011014342,0.0019035337,0.00058901595,0.00093289244,0.002273731,0.0020827318,0.00608343],"category_scores_gemma":[0.0027214796,0.00039515703,0.0015134894,0.0012200612,0.00070858473,0.002520978,0.0022983197,0.0018154433,0.0014930833],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041079844,0.00026764697,0.001439429,0.00044061293,0.0001623008,0.00029238002,0.00041432236,0.031505797,0.048442762,0.011773183,0.028579181,0.87627155],"study_design_scores_gemma":[0.00010170799,0.0002655714,0.0018501675,0.000051092116,0.00017378751,0.00040189057,0.000116131916,0.91671443,0.022845093,0.037532255,0.019899938,0.000048055364],"about_ca_topic_score_codex":0.017408064,"about_ca_topic_score_gemma":0.013216422,"teacher_disagreement_score":0.017408064,"about_ca_system_score_codex":0.0014285301,"about_ca_system_score_gemma":0.001189344,"threshold_uncertainty_score":0.03461349},"labels":[],"label_agreement":null},{"id":"W3098282109","doi":"","title":"Curiosity Based Exploration for Learning Terrain Models","year":2015,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Terrain; Perplexity; Computer science; Artificial intelligence; Discriminative model; Path (computing); Motion planning; Machine learning; Robot; Computer vision; Language model; Cartography; Geography","score_opus":0.09872959158289273,"score_gpt":0.3370356043554977,"score_spread":0.23830601277260494,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3098282109","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20704582,0.001315844,0.7851815,0.00091813673,0.000037128764,0.00006597963,0.0007353652,0.0016513016,0.0030488821],"genre_scores_gemma":[0.9373601,0.00037201832,0.059055936,0.00014051123,0.00005057237,0.00009007609,0.0011527822,0.00013170396,0.0016463138],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99956733,0.0001598157,0.00001768296,0.00015814778,0.000055191704,0.000041944448],"domain_scores_gemma":[0.99820685,0.0012436123,0.00018921382,0.00018668029,0.00009346411,0.00008027657],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008937875,0.00084933225,0.00087230693,0.0010520162,0.00043190914,0.0007326535,0.0010880776,0.00068226224,0.0020352006],"category_scores_gemma":[0.0047697457,0.00048926566,0.00097594823,0.00086019834,0.00078712124,0.0015237225,0.0012914902,0.0012524263,0.0004276624],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003236621,0.00008537349,0.008177442,0.00021281817,0.00022440488,0.00015396155,0.00046242963,0.8091917,0.0044757198,0.014571304,0.0035960828,0.15852502],"study_design_scores_gemma":[0.000008910411,0.00002365465,0.000482536,0.000009011034,0.000008043891,0.000029829449,0.000017465663,0.98544806,0.00050934026,0.013048927,0.00040808454,0.000006169019],"about_ca_topic_score_codex":0.0037366305,"about_ca_topic_score_gemma":0.0043440377,"teacher_disagreement_score":0.0037366305,"about_ca_system_score_codex":0.0008368858,"about_ca_system_score_gemma":0.00048239864,"threshold_uncertainty_score":0.0074297786},"labels":[],"label_agreement":null},{"id":"W3101703188","doi":"10.18653/v1/2020.findings-emnlp.44","title":"ConceptBert: Concept-Aware Representation for Visual Question Answering","year":2020,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":142,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Thales (Canada)","funders":"","keywords":"Question answering; Computer science; Embedding; Commonsense knowledge; Artificial intelligence; Exploit; Natural language processing; Natural language; Knowledge representation and reasoning; Focus (optics); Representation (politics); Information retrieval","score_opus":0.030750271106208177,"score_gpt":0.3548092531981962,"score_spread":0.324058982091988,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3101703188","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010073082,0.00088462525,0.95418805,0.00072414643,0.00018750058,0.00052256987,0.0037534342,0.025692116,0.0039744256],"genre_scores_gemma":[0.14263453,0.0005121476,0.8354788,0.00077729,0.00012224431,0.0007388682,0.013933113,0.00089357956,0.0049095093],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986022,0.00038706305,0.000067789144,0.00048371698,0.00034796065,0.00011130589],"domain_scores_gemma":[0.99819607,0.0008647682,0.00010109085,0.00045161048,0.0002809676,0.000105460465],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019195133,0.0015520835,0.00085052394,0.0022297981,0.00073370954,0.0019155863,0.0034723417,0.0030196155,0.011699238],"category_scores_gemma":[0.0072139767,0.00051315565,0.0014824154,0.0015521707,0.0010299751,0.006030306,0.0038646688,0.003094825,0.0036993814],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038697536,0.00046706904,0.0017932774,0.00087252836,0.00013675856,0.0001938276,0.0008493108,0.02888164,0.015043472,0.056641717,0.09834406,0.7963893],"study_design_scores_gemma":[0.00011306068,0.00020297426,0.001054009,0.00016678925,0.00006961024,0.0003987146,0.00037400017,0.7465235,0.021469852,0.16461965,0.06493892,0.00006886983],"about_ca_topic_score_codex":0.006910786,"about_ca_topic_score_gemma":0.009181192,"teacher_disagreement_score":0.011699238,"about_ca_system_score_codex":0.0017256753,"about_ca_system_score_gemma":0.0015764108,"threshold_uncertainty_score":0.03913784},"labels":[],"label_agreement":null},{"id":"W3105009590","doi":"","title":"Multimodal Graph Networks for Compositional Generalization in Visual Question Answering","year":2020,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Question answering; Generalization; Graph; Artificial intelligence; Graph theory; Natural language processing; Theoretical computer science; Mathematics; Combinatorics","score_opus":0.015426769876601897,"score_gpt":0.2875687495257047,"score_spread":0.2721419796491028,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3105009590","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038819067,0.0007361886,0.9526677,0.0008728802,0.00007138771,0.00010783489,0.0005721598,0.0017390439,0.004413774],"genre_scores_gemma":[0.71018136,0.00074667035,0.2779891,0.0004964327,0.0001396185,0.00028379358,0.0017986838,0.00042404895,0.0079404805],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995555,0.00017586914,0.000020302938,0.00016371481,0.000049583323,0.00003509552],"domain_scores_gemma":[0.9985233,0.0009367147,0.000089946516,0.00026064023,0.0001247239,0.000064631764],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007945563,0.00066821784,0.00074865774,0.0011004962,0.0006285821,0.0009948636,0.0013898119,0.0014295719,0.008518678],"category_scores_gemma":[0.005959865,0.0004129341,0.001106127,0.001088499,0.00095329696,0.003476007,0.0017961236,0.0017872908,0.0011880765],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042124907,0.0002690482,0.0017470531,0.00049868756,0.00016876892,0.00024829697,0.00083040324,0.23931396,0.018323427,0.30599338,0.014726819,0.41745886],"study_design_scores_gemma":[0.0000117238005,0.000021737884,0.00023780693,0.000019635283,0.000021782833,0.000022921344,0.000045971377,0.71636283,0.0012762176,0.28031546,0.0016543384,0.000009546802],"about_ca_topic_score_codex":0.0062097227,"about_ca_topic_score_gemma":0.008733471,"teacher_disagreement_score":0.008518678,"about_ca_system_score_codex":0.0010782861,"about_ca_system_score_gemma":0.0005430257,"threshold_uncertainty_score":0.028497756},"labels":[],"label_agreement":null},{"id":"W3106697459","doi":"10.18653/v1/2020.aacl-main.48","title":"Multimodal Pretraining for Dense Video Captioning","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Closed captioning; Timeline; Computer science; Leverage (statistics); Variety (cybernetics); Multimedia; Construct (python library); Artificial intelligence; Human–computer interaction; Natural language processing; Image (mathematics)","score_opus":0.044387721720699685,"score_gpt":0.32001766122166764,"score_spread":0.2756299395009679,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3106697459","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021665579,0.0032303652,0.93534416,0.0007079695,0.00084494957,0.00029725567,0.003099591,0.022363074,0.01244702],"genre_scores_gemma":[0.34868714,0.002124134,0.59356076,0.000994274,0.0009834132,0.00082424434,0.02265137,0.0025338577,0.027640818],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99928916,0.00018068712,0.000031559746,0.00023964336,0.00012689065,0.00013206045],"domain_scores_gemma":[0.99864024,0.00054640655,0.000056886034,0.00024815399,0.00042627723,0.00008204098],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011340993,0.0022338894,0.0011191,0.001721854,0.00077458017,0.0012164465,0.0017733283,0.0021606034,0.019964155],"category_scores_gemma":[0.0044752494,0.00092166674,0.001102547,0.0014988248,0.0006259218,0.0022758706,0.0019660774,0.0024238604,0.010166586],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00053207914,0.00023287509,0.0005369796,0.0003269749,0.00011501929,0.00020042337,0.00011994965,0.052876037,0.029402185,0.0042356635,0.07843085,0.83299094],"study_design_scores_gemma":[0.00004615558,0.0001595267,0.000998593,0.00008406571,0.00006148572,0.00014634874,0.00009172151,0.9516342,0.02381822,0.0088506015,0.014072834,0.000036129597],"about_ca_topic_score_codex":0.008827565,"about_ca_topic_score_gemma":0.018728461,"teacher_disagreement_score":0.019964155,"about_ca_system_score_codex":0.00097567297,"about_ca_system_score_gemma":0.0010187785,"threshold_uncertainty_score":0.066786766},"labels":[],"label_agreement":null},{"id":"W3108413361","doi":"10.1007/978-3-030-58583-9_22","title":"Structure-Aware Generation Network for Recipe Generation from Images","year":2020,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"BC Research (Canada)","funders":"","keywords":"Recipe; Computer science; Closed captioning; Task (project management); Benchmark (surveying); Tree (set theory); Artificial intelligence; Tree structure; Natural language processing; Sentence; Machine learning; Image (mathematics); Information retrieval; Data structure","score_opus":0.027145977666984025,"score_gpt":0.26912500691998603,"score_spread":0.24197902925300202,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3108413361","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01870496,0.00041050167,0.9579482,0.00022266545,0.00016496514,0.0001582809,0.00094892585,0.015825335,0.005616102],"genre_scores_gemma":[0.30859515,0.0004021568,0.66682976,0.00029298896,0.00010979283,0.00031999947,0.0048967367,0.0012043461,0.017349068],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99981004,0.000022317881,0.0000081493345,0.00009155244,0.000044699595,0.000023269997],"domain_scores_gemma":[0.9997445,0.00008669724,0.000013840993,0.00007164432,0.00006211627,0.000021313896],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00034724714,0.0010129382,0.00076489453,0.00081638014,0.0005120458,0.0006777905,0.0020176,0.0012001953,0.012106361],"category_scores_gemma":[0.00087912445,0.00060775114,0.0009080764,0.0006888299,0.00037877364,0.0010948083,0.0010940259,0.0012840048,0.003310932],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038765432,0.0003120856,0.0007781602,0.00017263176,0.00009511309,0.00021755413,0.00009126861,0.17590834,0.030547172,0.010816437,0.02261163,0.758062],"study_design_scores_gemma":[0.000014637074,0.000027090136,0.00014682785,0.0000066529647,0.000014723681,0.000042547857,0.000007809276,0.98557997,0.007321882,0.0047236397,0.0021054375,0.000008830068],"about_ca_topic_score_codex":0.0068050465,"about_ca_topic_score_gemma":0.0115604,"teacher_disagreement_score":0.012106361,"about_ca_system_score_codex":0.0007754708,"about_ca_system_score_gemma":0.0007216109,"threshold_uncertainty_score":0.040499806},"labels":[],"label_agreement":null},{"id":"W3111344731","doi":"10.1109/smc42975.2020.9283183","title":"Quantifying the Impact of Complementary Visual and Textual Cues Under Image Captioning","year":2020,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Lakehead University","funders":"","keywords":"Computer science; Closed captioning; Artificial intelligence; Feature (linguistics); Convolutional neural network; Encoder; Sentence; Natural language processing; Representation (politics); Sensory cue; Recurrent neural network; Pattern recognition (psychology); Image (mathematics); Speech recognition; Artificial neural network","score_opus":0.06289478724702684,"score_gpt":0.3833081381672803,"score_spread":0.32041335092025347,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3111344731","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.782808,0.0018176241,0.19788301,0.0004386621,0.0003972514,0.00038724518,0.00091279496,0.00481053,0.010544923],"genre_scores_gemma":[0.9338144,0.00043022362,0.061767887,0.00019447108,0.00006506676,0.00013933735,0.001046926,0.00051120773,0.0020303167],"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99872285,0.00031794354,0.00007869505,0.00027590583,0.00045948516,0.00014509325],"domain_scores_gemma":[0.9945838,0.0033547678,0.0004940865,0.0007287734,0.0006354045,0.00020329558],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001348185,0.0013900223,0.00050919823,0.0007615203,0.00034648774,0.0012337398,0.0010423746,0.00093918963,0.0031946383],"category_scores_gemma":[0.013873039,0.00031468208,0.00042643913,0.0005414738,0.00076635595,0.0027828298,0.0014524941,0.0010860364,0.00071381166],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025632605,0.0006945205,0.004155723,0.0013685633,0.00019834837,0.0010023542,0.00029356568,0.27605918,0.3996488,0.003470594,0.0036332149,0.30691195],"study_design_scores_gemma":[0.000050226325,0.0018383524,0.0046559027,0.000067913,0.00014559786,0.0006335059,0.00017426364,0.66474265,0.32235244,0.0025146585,0.0027524068,0.00007217418],"about_ca_topic_score_codex":0.0013083135,"about_ca_topic_score_gemma":0.0014167893,"teacher_disagreement_score":0.0031946383,"about_ca_system_score_codex":0.00065288093,"about_ca_system_score_gemma":0.00043655574,"threshold_uncertainty_score":0.010687113},"labels":[],"label_agreement":null},{"id":"W3111882491","doi":"","title":"MultiON: Benchmarking Semantic Map Memory using Multi-Object Navigation","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Task (project management); Artificial intelligence; Oracle; Mobile robot navigation; Computer vision; Observability; Turn-by-turn navigation; Feature (linguistics); Set (abstract data type); Navigation system; Human–computer interaction; Mobile robot; Robot","score_opus":0.1059553040641273,"score_gpt":0.23744483563010366,"score_spread":0.13148953156597637,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3111882491","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8177366,0.0031066414,0.12648995,0.00078770536,0.0009009177,0.00043790677,0.0062559014,0.02854669,0.015737614],"genre_scores_gemma":[0.8900209,0.00052500836,0.09265403,0.00033187546,0.00006380752,0.0002867662,0.009495582,0.0007784499,0.005843599],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994431,0.00013923082,0.000039399023,0.00021461885,0.00008830304,0.00007531617],"domain_scores_gemma":[0.9986754,0.0006086218,0.00008865019,0.00029484642,0.00019901185,0.00013331113],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011315272,0.0016047009,0.0007933162,0.0009276275,0.00050169503,0.0012616611,0.0026897613,0.002280362,0.005249102],"category_scores_gemma":[0.0042455383,0.0004605295,0.00080052816,0.00082197506,0.00067582674,0.0021502238,0.0017007426,0.0014916409,0.0019535357],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020178307,0.0017277692,0.014530495,0.001076394,0.0006849856,0.0003765666,0.00033152307,0.59026575,0.008226812,0.0037573942,0.028230194,0.34877434],"study_design_scores_gemma":[0.00012052973,0.00050364516,0.00180534,0.000031573876,0.00005137219,0.00006761015,0.00010802,0.9845057,0.006710618,0.003013775,0.0030494838,0.000032316282],"about_ca_topic_score_codex":0.020237574,"about_ca_topic_score_gemma":0.020598918,"teacher_disagreement_score":0.020237574,"about_ca_system_score_codex":0.0010276258,"about_ca_system_score_gemma":0.0012779216,"threshold_uncertainty_score":0.040239513},"labels":[],"label_agreement":null},{"id":"W3118480672","doi":"10.48550/arxiv.2101.01447","title":"End-to-End Video Question-Answer Generation with Generator-Pretester Network","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Ministry of Science and Technology, Taiwan; Nvidia","keywords":"Computer science; Task (project management); Generator (circuit theory); Question answering; Annotation; Artificial intelligence; Ground truth; Information retrieval; Multimedia; Power (physics)","score_opus":0.04454339780986381,"score_gpt":0.20328930949263468,"score_spread":0.15874591168277086,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3118480672","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.046158604,0.00083920424,0.892912,0.001021255,0.0003478581,0.0014313604,0.002855897,0.046373215,0.008060669],"genre_scores_gemma":[0.43061337,0.0003565178,0.5350643,0.0017243667,0.00021677438,0.0014392846,0.014027753,0.0012314899,0.015326134],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989292,0.00031874192,0.00004139125,0.00047230817,0.00013989572,0.00009839195],"domain_scores_gemma":[0.9980762,0.001091301,0.00008935388,0.00027269687,0.00035871833,0.000111727204],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014585019,0.0019333187,0.0009767481,0.00080002146,0.00048300577,0.00073500106,0.002334283,0.00228916,0.012978539],"category_scores_gemma":[0.0058570616,0.00040326736,0.00079963234,0.0004374841,0.0006382059,0.0021063236,0.0015813292,0.0018170236,0.0043269163],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012299108,0.0011395362,0.003261952,0.00071540114,0.00014967933,0.0012253898,0.0006671478,0.09435831,0.05889771,0.009079247,0.06875725,0.76051843],"study_design_scores_gemma":[0.00010758461,0.00027524566,0.0005488152,0.000028210161,0.000039457605,0.00023277313,0.000114158254,0.9484636,0.029779898,0.009319818,0.011061886,0.000028501456],"about_ca_topic_score_codex":0.004659517,"about_ca_topic_score_gemma":0.0052501056,"teacher_disagreement_score":0.012978539,"about_ca_system_score_codex":0.00108342,"about_ca_system_score_gemma":0.0011808572,"threshold_uncertainty_score":0.043417513},"labels":[],"label_agreement":null},{"id":"W3121592593","doi":"","title":"Long Range Arena : A Benchmark for Efficient Transformers","year":2021,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Suite; Benchmarking; Computer science; Transformer; Benchmark (surveying); Artificial intelligence; Machine learning; Data mining; Engineering","score_opus":0.01244723705163878,"score_gpt":0.2729188266178351,"score_spread":0.2604715895661963,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3121592593","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.39684734,0.019998772,0.4497386,0.005487196,0.0015943071,0.0012934706,0.033467513,0.04115406,0.050418727],"genre_scores_gemma":[0.65520525,0.0039322604,0.27558088,0.0011271815,0.00024188125,0.001005416,0.050078318,0.003429134,0.009399659],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997095,0.0010777181,0.0002773379,0.000750975,0.00057084084,0.0002280632],"domain_scores_gemma":[0.98969764,0.00684395,0.0003974066,0.0017418012,0.0010079959,0.0003112079],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042654243,0.002264989,0.0009533773,0.0020418453,0.00074036204,0.00228531,0.0036377334,0.0023489057,0.006817514],"category_scores_gemma":[0.02400835,0.0005741989,0.0017208561,0.0018453501,0.0012116712,0.0060216803,0.002200509,0.0024643547,0.0027533441],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014545147,0.00094160286,0.007846369,0.0034655181,0.0005638048,0.0004633311,0.00034260092,0.5112389,0.006682832,0.040289845,0.09339645,0.33331427],"study_design_scores_gemma":[0.0002944482,0.000965704,0.0012152612,0.00020544862,0.0001208178,0.0004012206,0.0002381987,0.9111115,0.008936742,0.05337664,0.023076179,0.000057929497],"about_ca_topic_score_codex":0.0068353014,"about_ca_topic_score_gemma":0.011430575,"teacher_disagreement_score":0.0068353014,"about_ca_system_score_codex":0.0019304594,"about_ca_system_score_gemma":0.0026515143,"threshold_uncertainty_score":0.022806883},"labels":[],"label_agreement":null},{"id":"W3122240496","doi":"10.1109/iccv48922.2021.01003","title":"Self-Supervised Representation Learning from Flow Equivariance","year":2021,"lang":"en","type":"article","venue":"2021 IEEE/CVF International Conference on Computer Vision (ICCV)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Artificial intelligence; Computer science; Representation (politics); Segmentation; Feature learning; Computer vision; Frame (networking); Object (grammar); Transformation (genetics); Transformation geometry; Pattern recognition (psychology)","score_opus":0.045308909399612,"score_gpt":0.3297275159253669,"score_spread":0.2844186065257549,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3122240496","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025421573,0.00033246062,0.9693768,0.000359773,0.00006082088,0.00008891208,0.00025868908,0.0027461336,0.0013547251],"genre_scores_gemma":[0.70024097,0.00044429564,0.2864882,0.0008437391,0.0003433904,0.00047859165,0.004199152,0.00072586915,0.006235788],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980849,0.00061123894,0.00008297376,0.0006681396,0.0003638926,0.00018898964],"domain_scores_gemma":[0.99537766,0.0020044628,0.00046455325,0.0010371824,0.00092776865,0.00018836543],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029137668,0.0018523723,0.0021178275,0.0016186706,0.00063325366,0.0014865156,0.0035509397,0.0023268356,0.0022278724],"category_scores_gemma":[0.008873594,0.00067673746,0.0013074751,0.0014489479,0.001687835,0.0034551532,0.002186397,0.0033729782,0.0012903664],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002936291,0.0003647849,0.0018774136,0.00017823812,0.00019636919,0.00011259695,0.00014341887,0.59575075,0.0045745294,0.016487977,0.018743938,0.3612763],"study_design_scores_gemma":[0.000010707686,0.000029092242,0.00007282816,0.0000049116243,0.0000058639316,0.000012862978,0.0000055205032,0.9928671,0.00074048794,0.0059726606,0.00027190094,0.0000059221325],"about_ca_topic_score_codex":0.0032852585,"about_ca_topic_score_gemma":0.0032846262,"teacher_disagreement_score":0.0035509397,"about_ca_system_score_codex":0.0015305774,"about_ca_system_score_gemma":0.0016423833,"threshold_uncertainty_score":0.015409708},"labels":[],"label_agreement":null},{"id":"W3122520957","doi":"10.1109/iros51168.2021.9636080","title":"Learning by Watching: Physical Imitation of Manipulation Skills from Human Videos","year":2021,"lang":"en","type":"article","venue":"2021 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Vector Institute; University of Toronto","funders":"","keywords":"Computer science; Artificial intelligence; Reinforcement learning; Robot; Imitation; Task (project management); Salient; Representation (politics); Unsupervised learning; Deep learning; Machine learning; Human–computer interaction","score_opus":0.0406571803570758,"score_gpt":0.3241998143792468,"score_spread":0.28354263402217095,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3122520957","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030215573,0.00020440907,0.96607053,0.00014484272,0.000026293003,0.000086086635,0.00010948347,0.0018260094,0.0013167098],"genre_scores_gemma":[0.7478623,0.00027498315,0.24765925,0.00018736385,0.000047262536,0.00023531953,0.000453504,0.00023399228,0.0030460323],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99960417,0.00010944486,0.0000160954,0.00014994074,0.00007878974,0.00004163136],"domain_scores_gemma":[0.9986332,0.00079150137,0.00017726424,0.00023438386,0.00008982577,0.00007370835],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007940158,0.00080767943,0.0006711182,0.00044237848,0.00024034388,0.0005592462,0.0015352776,0.0009257795,0.0015004992],"category_scores_gemma":[0.004666551,0.00048396992,0.0005115777,0.0003226859,0.00096758496,0.001253252,0.00090378476,0.0011542041,0.0003244807],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000267412,0.0002713365,0.0031390765,0.00024387643,0.00010106628,0.00033371523,0.00026411086,0.49073395,0.029651776,0.013087887,0.0036009862,0.45830485],"study_design_scores_gemma":[0.00001394732,0.000099689794,0.00046710548,0.000010192125,0.0000080815025,0.000057325484,0.000017124668,0.98719704,0.0052124714,0.0061445814,0.00076047675,0.000012080719],"about_ca_topic_score_codex":0.0041084834,"about_ca_topic_score_gemma":0.004675958,"teacher_disagreement_score":0.0041084834,"about_ca_system_score_codex":0.00061868096,"about_ca_system_score_gemma":0.0007998443,"threshold_uncertainty_score":0.008169174},"labels":[],"label_agreement":null},{"id":"W3133365413","doi":"10.1007/s10791-021-09398-0","title":"Neural ranking models for document retrieval","year":2021,"lang":"en","type":"preprint","venue":"Information Retrieval","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Instituto de Ciencias del Mar y Limnología, Universidad Nacional Autónoma de México; National Institute of Standards and Technology; Institute for Catastrophic Loss Reduction; National Science Foundation","keywords":"Ranking (information retrieval); Computer science; Artificial intelligence; Machine learning; Set (abstract data type); Variety (cybernetics); Information retrieval; Artificial neural network; Deep learning; Learning to rank; Data mining","score_opus":0.02256006568427007,"score_gpt":0.29165840351195466,"score_spread":0.2690983378276846,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3133365413","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.105906874,0.012281385,0.8390996,0.0043254434,0.000546144,0.00025319573,0.0015675529,0.00343275,0.03258703],"genre_scores_gemma":[0.8747833,0.0035065752,0.07991028,0.00059684133,0.0005817625,0.000323511,0.0014830643,0.00026308987,0.03855155],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992176,0.00029710119,0.000052529467,0.00014213286,0.00017828816,0.00011232211],"domain_scores_gemma":[0.9978795,0.0011181682,0.00024303254,0.0001667478,0.00050747395,0.00008515419],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020558096,0.00080945,0.0013602833,0.0019137101,0.00044864154,0.002023888,0.0016125607,0.0015463745,0.008365574],"category_scores_gemma":[0.006747213,0.00037738887,0.0008614809,0.0020979694,0.00065180194,0.0027603407,0.00067140226,0.0014695016,0.0027884739],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002744056,0.0002551485,0.0017188299,0.0003829457,0.00015284163,0.00014547426,0.0001144791,0.71367806,0.002390051,0.10800232,0.018563302,0.15432219],"study_design_scores_gemma":[0.000010149549,0.000021334905,0.00019924996,0.000011065277,0.000011085252,0.000017407894,0.000006081404,0.9778211,0.00015539832,0.020919751,0.0008187542,0.0000086821465],"about_ca_topic_score_codex":0.0074381493,"about_ca_topic_score_gemma":0.007762405,"teacher_disagreement_score":0.008365574,"about_ca_system_score_codex":0.0019865532,"about_ca_system_score_gemma":0.0007170533,"threshold_uncertainty_score":0.027985632},"labels":[],"label_agreement":null},{"id":"W3133424709","doi":"10.1109/iros45743.2020.9340905","title":"Understanding Contexts Inside Robot and Human Manipulation Tasks through Vision-Language Model and Ontology System in Video Streams","year":2020,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Ontology; Robot; Scheme (mathematics); Artificial intelligence; Human–computer interaction; Process (computing); Object (grammar); Mobile robot; Computer vision","score_opus":0.08615830094156307,"score_gpt":0.3290150866216928,"score_spread":0.24285678568012975,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3133424709","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.35684246,0.0008650151,0.63365436,0.0006894585,0.0001077693,0.00022270653,0.0013915444,0.0019879406,0.0042388495],"genre_scores_gemma":[0.89218074,0.00033622762,0.10381277,0.00012247938,0.000025107694,0.000094516734,0.001618522,0.00007568755,0.0017340791],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99961555,0.00007171397,0.000018341912,0.00016933409,0.00007208465,0.000052898034],"domain_scores_gemma":[0.9995937,0.00014356186,0.00005833241,0.00007515882,0.00008533293,0.000043863925],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00048480378,0.0004765002,0.00030222072,0.00084037555,0.00036135982,0.0010138405,0.0007471303,0.0007380381,0.0013013819],"category_scores_gemma":[0.0022573823,0.00019070022,0.0007503844,0.0006543644,0.0005336054,0.0022721922,0.0010170717,0.0009307186,0.00026129096],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011293258,0.00084398687,0.026918247,0.00042080885,0.0001963688,0.0012456198,0.0023086204,0.2499876,0.09413972,0.02928099,0.008367439,0.58516127],"study_design_scores_gemma":[0.000013742148,0.000064806896,0.008964447,0.000023771878,0.000040694045,0.00013852063,0.00040141837,0.96241796,0.013540294,0.011221221,0.0031427217,0.00003048134],"about_ca_topic_score_codex":0.030308131,"about_ca_topic_score_gemma":0.031863745,"teacher_disagreement_score":0.030308131,"about_ca_system_score_codex":0.0012946406,"about_ca_system_score_gemma":0.0010512333,"threshold_uncertainty_score":0.060263455},"labels":[],"label_agreement":null},{"id":"W3139837796","doi":"10.1016/j.media.2022.102374","title":"Weakly supervised segmentation with cross-modality equivariant constraints","year":2022,"lang":"en","type":"preprint","venue":"Medical Image Analysis","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"Compute Canada","keywords":"Computer science; Segmentation; Modality (human–computer interaction); Modalities; Artificial intelligence; Equivariant map; Machine learning; Constraint (computer-aided design); Exploit; Pattern recognition (psychology); Divergence (linguistics); Pixel; Mathematics","score_opus":0.016168797614745857,"score_gpt":0.34553033631595975,"score_spread":0.3293615387012139,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3139837796","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011706343,0.00022212318,0.9844296,0.00023464262,0.000045639677,0.000045277724,0.00027325345,0.001148399,0.0018947445],"genre_scores_gemma":[0.43835688,0.00062555843,0.54395336,0.00055834465,0.0004260888,0.00030270623,0.0029023737,0.0020761604,0.010798502],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99818414,0.00055219216,0.000106762884,0.00060828903,0.0003656404,0.00018295231],"domain_scores_gemma":[0.9973418,0.0010143752,0.0002933,0.00077848014,0.0003992019,0.00017281396],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018320654,0.0015080774,0.0022686245,0.0016706763,0.00071597507,0.0029801643,0.0022127526,0.002611321,0.005224087],"category_scores_gemma":[0.006611521,0.0012370211,0.0018890891,0.0020332334,0.0013525332,0.0025469395,0.0040094955,0.0029346852,0.002415689],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014088732,0.00034057623,0.0019914575,0.0006884213,0.0005054009,0.0007145763,0.0003021548,0.33573985,0.09138796,0.067342095,0.013322754,0.48625585],"study_design_scores_gemma":[0.000022822971,0.00006983536,0.00050597277,0.000027360858,0.000043466967,0.00019599452,0.000035968496,0.946338,0.013647418,0.03632512,0.0027615956,0.000026331583],"about_ca_topic_score_codex":0.0024347329,"about_ca_topic_score_gemma":0.004126684,"teacher_disagreement_score":0.005224087,"about_ca_system_score_codex":0.00075493497,"about_ca_system_score_gemma":0.0014148281,"threshold_uncertainty_score":0.01747626},"labels":[],"label_agreement":null},{"id":"W3157108758","doi":"10.1109/iccv48922.2021.01558","title":"Segmentation-grounded Scene Graph Generation","year":2021,"lang":"en","type":"preprint","venue":"2021 IEEE/CVF International Conference on Computer Vision (ICCV)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; U.S. Air Force; Compute Canada; Defense Advanced Research Projects Agency; Air Force Research Laboratory; Government of Canada; Canadian Institute for Advanced Research","keywords":"Computer science; Artificial intelligence; Scene graph; Segmentation; Graph; Pixel; Bounding overwatch; Pattern recognition (psychology); Granularity; Computer vision; Image segmentation; Object (grammar); Theoretical computer science","score_opus":0.06238528048595809,"score_gpt":0.3508478787254485,"score_spread":0.2884625982394904,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3157108758","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01266904,0.00019143255,0.9681193,0.00019030455,0.00007608948,0.00020399493,0.0015508863,0.01448023,0.0025186706],"genre_scores_gemma":[0.17356057,0.00019741942,0.8067406,0.00036063566,0.000052004933,0.00027907031,0.011098274,0.0031926367,0.0045187175],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99927706,0.00009574896,0.000020152393,0.00036072754,0.00016476955,0.00008166055],"domain_scores_gemma":[0.99920624,0.00021916983,0.00004578164,0.00029145172,0.00018088798,0.000056522753],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006807548,0.0015859328,0.0010152318,0.0014892409,0.0005952709,0.0012350413,0.002752759,0.0014875018,0.006668464],"category_scores_gemma":[0.0024645259,0.0007921046,0.0017125632,0.0012202819,0.0007424796,0.0020471253,0.0024739278,0.0018215774,0.0033476383],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005345383,0.00030153987,0.0034787767,0.00067037425,0.00022069397,0.00043831803,0.000545914,0.1767015,0.09402148,0.049383927,0.063092746,0.6106102],"study_design_scores_gemma":[0.00004275888,0.00007670606,0.00093476987,0.000027295722,0.000040499355,0.00020302217,0.00011342197,0.9051659,0.023689067,0.05556866,0.014102709,0.000035148252],"about_ca_topic_score_codex":0.004831817,"about_ca_topic_score_gemma":0.013299462,"teacher_disagreement_score":0.006668464,"about_ca_system_score_codex":0.0010371548,"about_ca_system_score_gemma":0.0013206191,"threshold_uncertainty_score":0.02230829},"labels":[],"label_agreement":null},{"id":"W3167565399","doi":"10.1109/tpami.2022.3194311","title":"NAAQA: A Neural Architecture for Acoustic Question Answering","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Pattern Analysis and Machine Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke","funders":"Natural Sciences and Engineering Research Council of Canada; CHIST-ERA; Nvidia","keywords":"Notation; Benchmark (surveying); Task (project management); Set (abstract data type); Question answering; Computer science; Artificial neural network; Artificial intelligence; Variable (mathematics); Architecture; Speech recognition; Mathematics; Arithmetic; Programming language; Engineering","score_opus":0.015824980517555244,"score_gpt":0.28340041613584366,"score_spread":0.2675754356182884,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3167565399","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19409522,0.0043692105,0.7336357,0.0017634066,0.0009961128,0.00078671105,0.009829716,0.037265148,0.017258773],"genre_scores_gemma":[0.619891,0.0010088644,0.33335373,0.0011753263,0.00016618022,0.0010700717,0.022430863,0.00065622333,0.020247685],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9995276,0.000110050816,0.000025231648,0.0001906032,0.00008836087,0.00005811068],"domain_scores_gemma":[0.99942243,0.00020174969,0.000029492194,0.00012130642,0.00017418085,0.000050801602],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00095479994,0.001254438,0.0005243971,0.000678602,0.0005560992,0.0010160581,0.0030501438,0.0018968397,0.006478496],"category_scores_gemma":[0.002903779,0.000563703,0.0010085064,0.00063649536,0.0005266864,0.0021431963,0.001913568,0.0021250057,0.0024627293],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008926878,0.0006300872,0.0061870962,0.0009121868,0.0005325348,0.0002530492,0.00033239444,0.2554763,0.038683183,0.009873689,0.050397314,0.6358294],"study_design_scores_gemma":[0.000060861617,0.00021299814,0.0011115227,0.000034364584,0.000051642535,0.00006905357,0.000057219633,0.97447395,0.0076923124,0.0076815686,0.00852797,0.000026465143],"about_ca_topic_score_codex":0.013191588,"about_ca_topic_score_gemma":0.018978475,"teacher_disagreement_score":0.013191588,"about_ca_system_score_codex":0.0010599135,"about_ca_system_score_gemma":0.0011841113,"threshold_uncertainty_score":0.02622962},"labels":[],"label_agreement":null},{"id":"W3171547673","doi":"10.48550/arxiv.2106.03089","title":"Referring Transformer: A One-step Approach to Multi-task Visual Grounding","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":73,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Ground; Transformer; Task (project management); Computer science; Engineering; Electrical engineering; Systems engineering; Voltage","score_opus":0.11713264509518619,"score_gpt":0.23845911190050736,"score_spread":0.12132646680532118,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3171547673","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0126248915,0.00048781128,0.9666355,0.0003600203,0.000091232665,0.00017238318,0.00032453923,0.015981194,0.0033224337],"genre_scores_gemma":[0.46074492,0.00053462206,0.5201771,0.0009398941,0.00017800245,0.0003539223,0.0018545194,0.0014266048,0.013790499],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988501,0.00027362446,0.00004359276,0.0004848782,0.0001836299,0.00016413718],"domain_scores_gemma":[0.99895954,0.00033810022,0.000074436764,0.00037436653,0.00016896591,0.00008460917],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017223901,0.001972593,0.0013595984,0.0010304742,0.0004758947,0.0015915238,0.0041389335,0.0023919882,0.008347004],"category_scores_gemma":[0.0037033185,0.0007253333,0.0020775876,0.0011000135,0.0011079753,0.0039022202,0.0036510013,0.002805115,0.0034471818],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055427226,0.000396482,0.0009274456,0.00031148124,0.000213932,0.00039537455,0.0004915165,0.08682006,0.049955547,0.018350028,0.016296355,0.82528746],"study_design_scores_gemma":[0.000043547894,0.00024357544,0.0005558015,0.000028438544,0.00008521201,0.0001981535,0.00010125218,0.93503505,0.02479108,0.032237507,0.0066311685,0.000049116177],"about_ca_topic_score_codex":0.0051095337,"about_ca_topic_score_gemma":0.0069595794,"teacher_disagreement_score":0.008347004,"about_ca_system_score_codex":0.0009918535,"about_ca_system_score_gemma":0.0013014073,"threshold_uncertainty_score":0.027923524},"labels":[],"label_agreement":null},{"id":"W3172726328","doi":"10.18653/v1/2021.naacl-main.153","title":"Worldly Wise (WoW) - Cross-Lingual Knowledge Fusion for Fact-based Visual Spoken-Question Answering","year":2021,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Question answering; Computer science; Natural language processing; Fusion; Artificial intelligence; Information retrieval; Linguistics; Philosophy","score_opus":0.018629945614470084,"score_gpt":0.37460044306566087,"score_spread":0.3559704974511908,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3172726328","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033387996,0.0025378373,0.8600829,0.0015889492,0.00095487863,0.00090167753,0.01477614,0.070967495,0.014802127],"genre_scores_gemma":[0.26661542,0.0008482758,0.6711423,0.000907899,0.00022234322,0.00087374094,0.04557463,0.0021217223,0.011693691],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978459,0.00046850045,0.00022784893,0.00074205163,0.00044794354,0.00026784383],"domain_scores_gemma":[0.99769235,0.00075988454,0.00008929533,0.00075587607,0.00054494623,0.00015762022],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033993023,0.0018770876,0.0015834747,0.0044568367,0.0013903811,0.0033270274,0.0026150106,0.002712451,0.018353054],"category_scores_gemma":[0.0071722283,0.00081261474,0.0020038616,0.0027320306,0.0008137407,0.009428369,0.010907048,0.0023009332,0.011616295],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007031283,0.0007087869,0.0016729287,0.0003962844,0.0003511575,0.00045060928,0.0010913801,0.0076110302,0.013936094,0.012958329,0.088865474,0.87125486],"study_design_scores_gemma":[0.00021843555,0.0005045514,0.0048468267,0.00026004773,0.0004047412,0.00045704155,0.0035824303,0.7309249,0.037844032,0.12788433,0.0928335,0.00023926751],"about_ca_topic_score_codex":0.015165658,"about_ca_topic_score_gemma":0.022751972,"teacher_disagreement_score":0.018353054,"about_ca_system_score_codex":0.0009579066,"about_ca_system_score_gemma":0.0018302945,"threshold_uncertainty_score":0.061397076},"labels":[],"label_agreement":null},{"id":"W3173904629","doi":"10.21428/594757db.898c7976","title":"Descriptive Image Captioning with Salient Retrieval Priors","year":2021,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Salient; Closed captioning; Benchmark (surveying); Prior probability; Exploit; Artificial intelligence; Image (mathematics); Range (aeronautics); Modal; Information retrieval; Natural language processing; Pattern recognition (psychology); Machine learning; Bayesian probability","score_opus":0.009867451841899348,"score_gpt":0.2439625610612905,"score_spread":0.23409510921939114,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3173904629","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04532915,0.006454003,0.90526557,0.0020300008,0.0006939187,0.00089193153,0.0028050207,0.015905337,0.020625062],"genre_scores_gemma":[0.48478436,0.003678404,0.46964914,0.00200901,0.0013150874,0.0006873887,0.010263551,0.0013904345,0.02622258],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99909806,0.0002588916,0.00005963906,0.00032172818,0.00018712907,0.000074517266],"domain_scores_gemma":[0.9975344,0.0009137487,0.00028690777,0.00058800465,0.0005800114,0.000096769196],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014477383,0.0017168827,0.0008419064,0.0017955252,0.00045529526,0.0016323328,0.0015815271,0.0018457735,0.005687757],"category_scores_gemma":[0.0084232865,0.00038977974,0.0008488395,0.0013340934,0.0009085416,0.0039681667,0.0012660675,0.001793874,0.0045336066],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006608406,0.000360237,0.001258623,0.0010105274,0.00010206686,0.0003529518,0.00037858152,0.0839678,0.04470207,0.014090352,0.0684451,0.7846708],"study_design_scores_gemma":[0.000068577094,0.00036073805,0.0013846199,0.00013469708,0.0001147431,0.0006966524,0.00013311158,0.9062562,0.03381853,0.021575985,0.035372656,0.00008351298],"about_ca_topic_score_codex":0.0022076564,"about_ca_topic_score_gemma":0.0034957363,"teacher_disagreement_score":0.005687757,"about_ca_system_score_codex":0.0011151752,"about_ca_system_score_gemma":0.00077638327,"threshold_uncertainty_score":0.019027412},"labels":[],"label_agreement":null},{"id":"W3173937618","doi":"10.18653/v1/2021.acl-short.36","title":"Enhancing Descriptive Image Captioning with Natural Language Inference","year":2021,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Closed captioning; Computer science; Inference; Natural language; Computational linguistics; Natural language processing; Natural (archaeology); Joint (building); Artificial intelligence; Linguistics; Image (mathematics); History; Engineering; Philosophy","score_opus":0.007255126184124819,"score_gpt":0.2684051231919964,"score_spread":0.26114999700787156,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3173937618","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009457507,0.0022874794,0.9549755,0.00090706506,0.00094611745,0.0002187301,0.002163227,0.019277744,0.00976671],"genre_scores_gemma":[0.16146305,0.0025946454,0.81054777,0.00088900176,0.0006991553,0.00026406726,0.0092735635,0.00276366,0.01150506],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991691,0.0002556214,0.00005358897,0.00022212394,0.00023135485,0.00006822453],"domain_scores_gemma":[0.9978922,0.00085255667,0.00011984335,0.0004640321,0.00061296247,0.000058411384],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012010079,0.0015972272,0.0009861393,0.0019976175,0.0006909252,0.0022351737,0.0018147739,0.0014091737,0.011328714],"category_scores_gemma":[0.0048424765,0.00060121046,0.0013225053,0.0017736143,0.0007722728,0.0044185943,0.0017808175,0.0022425724,0.0052904678],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036227587,0.00026665156,0.00040401073,0.0011283703,0.000109558045,0.00036467475,0.00037534107,0.022412559,0.07624724,0.02676739,0.128841,0.7427209],"study_design_scores_gemma":[0.000081839826,0.000118123666,0.00062552944,0.00008699948,0.00015752444,0.0004964765,0.00022956313,0.7821003,0.09908224,0.03798409,0.07895012,0.00008720598],"about_ca_topic_score_codex":0.003043111,"about_ca_topic_score_gemma":0.005256183,"teacher_disagreement_score":0.011328714,"about_ca_system_score_codex":0.00067848753,"about_ca_system_score_gemma":0.0007906469,"threshold_uncertainty_score":0.03789836},"labels":[],"label_agreement":null},{"id":"W3174481471","doi":"10.18653/v1/2021.acl-long.564","title":"Mind Your Outliers! Investigating the Negative Impact of Outliers on Active Learning for Visual Question Answering","year":2021,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":50,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Open Philanthropy Project; Canadian Institute for Advanced Research","keywords":"Question answering; Outlier; Computational linguistics; Natural language processing; Computer science; Artificial intelligence; Corpus linguistics; Volume (thermodynamics); Linguistics; Information retrieval; Philosophy","score_opus":0.029758105145624873,"score_gpt":0.36747120422434215,"score_spread":0.33771309907871727,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3174481471","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6390908,0.01693474,0.30310297,0.01672932,0.0020233844,0.00027805267,0.0027566883,0.005689125,0.013394977],"genre_scores_gemma":[0.9548419,0.0008501634,0.037522905,0.0009493959,0.00049485615,0.00009593361,0.00169161,0.00036696074,0.0031862773],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9948037,0.0028135877,0.00016979566,0.00083064946,0.0011917627,0.000190503],"domain_scores_gemma":[0.9137367,0.07272369,0.0025309215,0.0044778413,0.005426312,0.0011044878],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009100189,0.0007419929,0.0011080961,0.00078640593,0.0010376574,0.0022840698,0.0020517118,0.0024027296,0.0031948236],"category_scores_gemma":[0.098843314,0.00045561613,0.00047906634,0.00079692504,0.0011742254,0.005813006,0.002444939,0.003218982,0.0015958804],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005777064,0.0015107453,0.050051257,0.0010159257,0.0004802217,0.00048362315,0.0031923056,0.054164715,0.028274069,0.018408623,0.09949065,0.73715085],"study_design_scores_gemma":[0.00021060687,0.00083741714,0.016505161,0.00018570542,0.00015929068,0.00037665715,0.0017880077,0.88246137,0.017367711,0.06480104,0.015232143,0.00007497898],"about_ca_topic_score_codex":0.0034467361,"about_ca_topic_score_gemma":0.0036427472,"teacher_disagreement_score":0.009100189,"about_ca_system_score_codex":0.00056168565,"about_ca_system_score_gemma":0.00048856984,"threshold_uncertainty_score":0.048126936},"labels":[],"label_agreement":null},{"id":"W3176425931","doi":"10.1609/aaai.v35i3.16353","title":"Semantic Grouping Network for Video Captioning","year":2021,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":148,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Computer science; Interpretability; Closed captioning; Margin (machine learning); Word (group theory); Redundancy (engineering); Phrase; Artificial intelligence; Natural language processing; Speech recognition; Image (mathematics); Machine learning","score_opus":0.06834906582068068,"score_gpt":0.31396189647539346,"score_spread":0.2456128306547128,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3176425931","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.048060298,0.0029064948,0.9294325,0.0006551074,0.00039939347,0.00038807536,0.0013315782,0.0079670325,0.008859577],"genre_scores_gemma":[0.57201034,0.0017600555,0.405378,0.0005890199,0.00039951853,0.00045516537,0.0060517294,0.00049671927,0.012859441],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993431,0.00018330087,0.00003164765,0.00021790592,0.00014092779,0.000083102925],"domain_scores_gemma":[0.9992931,0.0002443895,0.00009670693,0.000113574635,0.00019680013,0.00005538766],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011218971,0.001958558,0.0010002828,0.0023765315,0.00069407723,0.000765211,0.0019194402,0.0016548525,0.004047051],"category_scores_gemma":[0.0033610142,0.0004003341,0.000961724,0.0020863484,0.0007031387,0.0024392933,0.0011934203,0.0012848184,0.0015784828],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00076754234,0.00025286313,0.0015283506,0.00037212844,0.00013110736,0.00038992584,0.00032107113,0.2122558,0.024381498,0.009351962,0.018160919,0.7320869],"study_design_scores_gemma":[0.000019705627,0.00014063186,0.00052651,0.00003374656,0.00006076972,0.00012457333,0.00008453285,0.97081876,0.0106593575,0.010629371,0.0068742023,0.000027948801],"about_ca_topic_score_codex":0.0066175237,"about_ca_topic_score_gemma":0.0061041047,"teacher_disagreement_score":0.0066175237,"about_ca_system_score_codex":0.0015446226,"about_ca_system_score_gemma":0.00076652865,"threshold_uncertainty_score":0.013538778},"labels":[],"label_agreement":null},{"id":"W3177283914","doi":"10.1145/3452918.3458795","title":"Context-Aware Question-Answer for Interactive Media Experiences","year":2021,"lang":"en","type":"article","venue":"ACM International Conference on Interactive Media Experiences","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Context (archaeology); Entertainment; Quality (philosophy); Multimedia; World Wide Web; Information retrieval; Context model; Artificial intelligence","score_opus":0.05343325353059438,"score_gpt":0.3770970106831698,"score_spread":0.32366375715257545,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3177283914","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038850196,0.001902193,0.9427286,0.00049318856,0.00008945582,0.00066335016,0.00051448477,0.00983394,0.00492465],"genre_scores_gemma":[0.4669585,0.0006655571,0.5263633,0.00038275865,0.00008271952,0.0006003723,0.0014112935,0.00029100533,0.0032444212],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99845326,0.00068888,0.00011261467,0.00030641397,0.0003322446,0.0001066082],"domain_scores_gemma":[0.997875,0.0012907285,0.00012733348,0.00026984944,0.0002990932,0.00013808928],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017657914,0.0008617678,0.00050336844,0.0009400952,0.0005176927,0.0013601214,0.0014656361,0.0014259267,0.0063801897],"category_scores_gemma":[0.008246648,0.0002785531,0.0006617952,0.00047306024,0.00054667366,0.00321164,0.002281063,0.0011605859,0.0014516425],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011247356,0.0008879962,0.0036483945,0.0025629986,0.00018016942,0.0006691413,0.0055180555,0.031209992,0.07436222,0.03097712,0.015889257,0.83297],"study_design_scores_gemma":[0.00026484812,0.00097235397,0.0066266675,0.0004319688,0.00032977673,0.0010249191,0.0023580592,0.7213942,0.054599326,0.09282776,0.11897284,0.00019735975],"about_ca_topic_score_codex":0.0018246371,"about_ca_topic_score_gemma":0.0025243028,"teacher_disagreement_score":0.0063801897,"about_ca_system_score_codex":0.00042983575,"about_ca_system_score_gemma":0.00042092957,"threshold_uncertainty_score":0.021343887},"labels":[],"label_agreement":null},{"id":"W3192616105","doi":"10.48550/arxiv.2108.03353","title":"Screen2Words: Automatic Mobile UI Summarization with Multimodal Learning","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Automatic summarization; Computer science; Bridging (networking); Phrase; Artificial intelligence; Modal; Semantics (computer science); Natural language processing; Mobile device; Set (abstract data type); Information retrieval; Human–computer interaction; World Wide Web; Programming language","score_opus":0.028257931777442832,"score_gpt":0.19122617935735842,"score_spread":0.16296824757991557,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3192616105","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.110196136,0.003995143,0.62379605,0.0007678505,0.00046221033,0.0012566845,0.055073917,0.19445607,0.009995971],"genre_scores_gemma":[0.30443722,0.001048435,0.51197106,0.0006479426,0.00018348046,0.0015682164,0.1601854,0.0027189602,0.017239274],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992719,0.00015664869,0.000059744914,0.0002551978,0.00017957292,0.00007691023],"domain_scores_gemma":[0.99907386,0.00028296193,0.00008467206,0.00020321246,0.0002989957,0.000056283014],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00064065575,0.0026812037,0.0007747879,0.002564242,0.00045152902,0.0010433701,0.0017716037,0.0012427897,0.005375295],"category_scores_gemma":[0.0036072293,0.00033504813,0.0011705394,0.00121656,0.00030629183,0.0017816095,0.0016843227,0.0012787812,0.0051038675],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00081646256,0.000470802,0.004668708,0.0013482731,0.00027445093,0.00036010277,0.00045224416,0.023390438,0.048736498,0.0017570347,0.12465205,0.7930729],"study_design_scores_gemma":[0.00021150502,0.00087918993,0.009512015,0.0002325883,0.00025396797,0.0004425983,0.0008131673,0.80491,0.09026818,0.008581564,0.08372782,0.0001673179],"about_ca_topic_score_codex":0.007319795,"about_ca_topic_score_gemma":0.016440224,"teacher_disagreement_score":0.007319795,"about_ca_system_score_codex":0.0009598411,"about_ca_system_score_gemma":0.00075472886,"threshold_uncertainty_score":0.017982185},"labels":[],"label_agreement":null},{"id":"W3195852121","doi":"10.1109/ecbios51820.2021.9510291","title":"Policy and Value Deep RL for Temporal Language-Agnostic Street Image Captioning","year":2021,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Closed captioning; Computer science; Artificial intelligence; Natural language; Image (mathematics); Word (group theory); Natural language processing; Encoder; Linguistics","score_opus":0.009107360291103987,"score_gpt":0.30280202643381243,"score_spread":0.29369466614270845,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3195852121","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.047728084,0.000766208,0.94127405,0.00076602347,0.00015697604,0.00012368445,0.0004423633,0.004590197,0.0041524735],"genre_scores_gemma":[0.8102828,0.00032369976,0.18053298,0.00075836026,0.00011960181,0.0002292299,0.0011503898,0.00035820866,0.006244685],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994628,0.00017979521,0.00002203276,0.00018323437,0.000070311995,0.0000818223],"domain_scores_gemma":[0.99903893,0.0005630668,0.00008786243,0.00011048441,0.00013630376,0.00006345688],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013066044,0.00094955094,0.0008563668,0.0004622866,0.00035299797,0.00084310287,0.0014837196,0.0012596427,0.0027858168],"category_scores_gemma":[0.004580143,0.0003876315,0.0005431682,0.0005603277,0.0009314558,0.0016382954,0.0010479265,0.0020347123,0.0008648101],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031396133,0.0002476131,0.0013796946,0.00017332175,0.00007731488,0.00019749883,0.00018198158,0.690503,0.011205077,0.010464904,0.010065473,0.27519014],"study_design_scores_gemma":[0.000008830915,0.000021224505,0.0000629504,0.0000052054966,0.000004494451,0.000011638402,0.00000929477,0.99368095,0.0017891875,0.003928429,0.0004726003,0.0000052085857],"about_ca_topic_score_codex":0.00702785,"about_ca_topic_score_gemma":0.008191028,"teacher_disagreement_score":0.00702785,"about_ca_system_score_codex":0.001444042,"about_ca_system_score_gemma":0.0013154492,"threshold_uncertainty_score":0.013973892},"labels":[],"label_agreement":null},{"id":"W3197581510","doi":"10.1609/aaai.v36i10.21306","title":"Retrieve, Caption, Generate: Visual Grounding for Enhancing Commonsense in Text Generation Models","year":2022,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Commonsense reasoning; Closed captioning; Computer science; Fluency; Transformer; Commonsense knowledge; Generative grammar; Artificial intelligence; Natural language processing; Image (mathematics); Linguistics; Engineering; Knowledge extraction","score_opus":0.14225812018072292,"score_gpt":0.355538293976938,"score_spread":0.21328017379621506,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3197581510","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0639674,0.0003749466,0.8950602,0.00070105185,0.00014972046,0.000375411,0.0012770433,0.026275108,0.01181908],"genre_scores_gemma":[0.546347,0.00019165024,0.44422346,0.00033235952,0.00005413192,0.00025086076,0.0025970975,0.0017165602,0.004286891],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994905,0.00020583704,0.000022940903,0.00013836006,0.000105896135,0.000036311787],"domain_scores_gemma":[0.9975936,0.0014932082,0.000108707005,0.0005459148,0.0001797743,0.00007880077],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013963026,0.0011098958,0.0004023778,0.0009306184,0.00033361954,0.001447661,0.0015110457,0.0011145974,0.009763787],"category_scores_gemma":[0.007630596,0.00035376492,0.0007883688,0.00044507984,0.0008111649,0.0031746423,0.0017509143,0.0013877334,0.0020080646],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006743526,0.00054421107,0.002388068,0.0007461716,0.000114208684,0.0006642825,0.0016763039,0.20927805,0.052570254,0.051779725,0.021898152,0.65766615],"study_design_scores_gemma":[0.00007672538,0.00016639703,0.0003832192,0.000042008494,0.00004611166,0.00020595377,0.0001560229,0.9070722,0.04190309,0.035865646,0.0140478015,0.000034911824],"about_ca_topic_score_codex":0.001896586,"about_ca_topic_score_gemma":0.003145586,"teacher_disagreement_score":0.009763787,"about_ca_system_score_codex":0.00065090926,"about_ca_system_score_gemma":0.0004935744,"threshold_uncertainty_score":0.032663107},"labels":[],"label_agreement":null},{"id":"W3200042178","doi":"10.33552/gjes.2020.04.000596","title":"SRIN: A New Dataset for Social Robot Indoor Navigation","year":2020,"lang":"en","type":"article","venue":"Global Journal of Engineering Sciences","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Western Canada Research Grid; Compute Canada","keywords":"The Internet; Computer science; Robot; Internet of Things; Computer security; Transport engineering; World Wide Web; Artificial intelligence; Engineering","score_opus":0.035873728696510024,"score_gpt":0.31969755222435503,"score_spread":0.283823823527845,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3200042178","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016862314,0.0019014651,0.0073947613,0.00046611318,0.00067994767,0.00037682758,0.9518922,0.010008543,0.010417853],"genre_scores_gemma":[0.015949434,0.00034012768,0.008614542,0.0001706833,0.000047490157,0.00038392044,0.97149354,0.0001769603,0.0028232492],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9988306,0.0002141172,0.00010291205,0.00037496688,0.00032636366,0.00015107973],"domain_scores_gemma":[0.9991959,0.0001048807,0.00006791669,0.00025165526,0.0002819752,0.00009780106],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005146941,0.0030254063,0.0016919055,0.0024884522,0.0014154452,0.000996666,0.003021867,0.0026251888,0.011929851],"category_scores_gemma":[0.0022909543,0.0004694432,0.0019235655,0.0031222934,0.00046506987,0.0012793294,0.0021790068,0.0018896573,0.023775538],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055028923,0.0004133172,0.010211837,0.0018362926,0.0003301551,0.0004664907,0.00021352449,0.0052580046,0.0025355585,0.0015282677,0.8927517,0.08390461],"study_design_scores_gemma":[0.00038951045,0.0005078905,0.04333891,0.0008147071,0.0002832468,0.0012392187,0.0012399676,0.036245696,0.00776609,0.004392082,0.9034656,0.00031708225],"about_ca_topic_score_codex":0.059670135,"about_ca_topic_score_gemma":0.13584551,"teacher_disagreement_score":0.059670135,"about_ca_system_score_codex":0.0011615737,"about_ca_system_score_gemma":0.0019085974,"threshold_uncertainty_score":0.11864567},"labels":[],"label_agreement":null},{"id":"W3200172683","doi":"10.18653/v1/2021.emnlp-main.516","title":"Mind the Context: The Impact of Contextualization in Neural Module Networks for Grounding Visual Referring Expressions","year":2021,"lang":"en","type":"article","venue":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Contextualization; Computer science; Generalization; Context (archaeology); Set (abstract data type); Exploit; Parameterized complexity; Artificial intelligence; Test set; Embedding; Cube (algebra); Machine learning; Algorithm; Mathematics; Programming language","score_opus":0.06160955148695172,"score_gpt":0.46076189392771477,"score_spread":0.39915234244076303,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3200172683","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34488717,0.0040735505,0.6210233,0.0019684825,0.00033325836,0.00018598164,0.0009781845,0.012462241,0.014087887],"genre_scores_gemma":[0.882308,0.00056310056,0.11038811,0.0007483577,0.00011304859,0.0001387635,0.001403192,0.00044452705,0.003892906],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99931526,0.00025700303,0.000025720545,0.00028777297,0.000048246173,0.00006603645],"domain_scores_gemma":[0.99922013,0.000393972,0.00007566831,0.0001666888,0.000098380915,0.000045184384],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010788332,0.0014869269,0.00057892955,0.00038385327,0.00048575416,0.0011220783,0.0019974483,0.0013547597,0.0034073584],"category_scores_gemma":[0.0044407374,0.0005083755,0.0009924402,0.00047480655,0.00069609267,0.0036504483,0.0016880729,0.002028863,0.00092007074],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009555078,0.0003321617,0.008352417,0.00046030956,0.00035089912,0.0005309218,0.0007072735,0.42765597,0.038551174,0.02106779,0.013649513,0.487386],"study_design_scores_gemma":[0.000033627915,0.0001242596,0.0011898692,0.000051446954,0.00010313062,0.00010275706,0.00007210169,0.9757529,0.0063278093,0.011971145,0.0042454205,0.000025547804],"about_ca_topic_score_codex":0.0051611285,"about_ca_topic_score_gemma":0.011362897,"teacher_disagreement_score":0.0051611285,"about_ca_system_score_codex":0.0010249707,"about_ca_system_score_gemma":0.0006630484,"threshold_uncertainty_score":0.011398733},"labels":[],"label_agreement":null},{"id":"W3202332982","doi":"10.1609/aaai.v36i10.21306","title":"Retrieve, Caption, Generate: Visual Grounding for Enhancing Commonsense in Text Generation Models","year":2022,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Closed captioning; Commonsense reasoning; Computer science; Fluency; Transformer; Generative grammar; Commonsense knowledge; Artificial intelligence; Natural language processing; Image (mathematics); Linguistics; Knowledge extraction; Engineering","score_opus":0.04270963079109104,"score_gpt":0.30356759393094573,"score_spread":0.2608579631398547,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3202332982","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06916484,0.00041862388,0.8836517,0.0007041393,0.00017739873,0.00041573073,0.0014369612,0.031439926,0.012590647],"genre_scores_gemma":[0.5603633,0.00019753077,0.42952007,0.00035875847,0.000057123936,0.00026498843,0.0028564285,0.0018587002,0.0045230663],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995271,0.0001828818,0.000021463002,0.00013498214,0.00009863231,0.000034914858],"domain_scores_gemma":[0.997759,0.0014060461,0.000100178615,0.00048543318,0.000175208,0.00007418661],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013465714,0.0011662835,0.00039977446,0.0009052554,0.0003321483,0.0013844373,0.0015084688,0.0010944476,0.010053203],"category_scores_gemma":[0.0074081877,0.0003445507,0.0007654217,0.00041193506,0.0007566134,0.0029736625,0.0016918089,0.0013882635,0.0020752747],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00067894696,0.0005634501,0.002561894,0.00075771956,0.00011819073,0.0006854215,0.001521691,0.2048738,0.053735476,0.042322535,0.023479735,0.66870123],"study_design_scores_gemma":[0.00007463121,0.00017404085,0.00040531906,0.00004317791,0.00004649406,0.00021233402,0.00015165324,0.9127682,0.043066457,0.028974716,0.014047313,0.0000357411],"about_ca_topic_score_codex":0.0019756877,"about_ca_topic_score_gemma":0.0033190586,"teacher_disagreement_score":0.010053203,"about_ca_system_score_codex":0.0006270796,"about_ca_system_score_gemma":0.00048660152,"threshold_uncertainty_score":0.033631384},"labels":[],"label_agreement":null},{"id":"W3203338521","doi":"10.18653/v1/2021.emnlp-main.328","title":"Language-Aligned Waypoint (LAW) Supervision for Vision-and-Language Navigation in Continuous Environments","year":2021,"lang":"en","type":"article","venue":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Western Canada Research Grid; Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Waypoint; Computer science; Task (project management); Human–computer interaction; Embodied cognition; Path (computing); Natural language; Work (physics); Measure (data warehouse); Metric (unit); Shortest path problem; Artificial intelligence; Multimedia; Programming language; Real-time computing; Engineering; Theoretical computer science; Database","score_opus":0.026221236821911142,"score_gpt":0.3976567197951307,"score_spread":0.37143548297321954,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3203338521","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05719918,0.0005740678,0.9377365,0.00016030275,0.00006926459,0.00008649895,0.00018152021,0.0029981532,0.0009946027],"genre_scores_gemma":[0.8040998,0.00024979774,0.1928853,0.00013635113,0.00006686397,0.00015001885,0.0007980163,0.00026179975,0.0013519658],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986266,0.00040618813,0.000086078115,0.00043145,0.00033949816,0.00011020735],"domain_scores_gemma":[0.9945878,0.0027768435,0.0006443874,0.0007009289,0.0010105473,0.00027956828],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018532525,0.00075415586,0.0008261462,0.0006477664,0.0005494103,0.0006954458,0.001807616,0.0010911318,0.0015866841],"category_scores_gemma":[0.013887045,0.00040094234,0.00048577428,0.00055277994,0.0011934999,0.0034524915,0.0019172729,0.0017709639,0.00042993922],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010545158,0.00051933114,0.008880778,0.00058774673,0.00013938724,0.00027305106,0.0010587228,0.29672167,0.022810286,0.015727833,0.0071546175,0.6450721],"study_design_scores_gemma":[0.00004525778,0.00030804652,0.0017929305,0.00002693252,0.00002323527,0.00008933361,0.00009969287,0.97357696,0.006775008,0.015802696,0.0014294898,0.000030367803],"about_ca_topic_score_codex":0.010746907,"about_ca_topic_score_gemma":0.014773388,"teacher_disagreement_score":0.010746907,"about_ca_system_score_codex":0.0007616959,"about_ca_system_score_gemma":0.0019461178,"threshold_uncertainty_score":0.021368682},"labels":[],"label_agreement":null},{"id":"W3204859046","doi":"10.18280/ts.380403","title":"Label Importance Ranking with Entropy Variation Complex Networks for Structured Video Captioning","year":2021,"lang":"en","type":"article","venue":"Traitement du signal","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Beijing Municipal Science and Technology Commission","keywords":"Variation (astronomy); Closed captioning; Ranking (information retrieval); Computer science; Entropy (arrow of time); Artificial intelligence; Data mining; Image (mathematics); Astrophysics","score_opus":0.019463123209371765,"score_gpt":0.2584458161975389,"score_spread":0.23898269298816716,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3204859046","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030607628,0.0008344271,0.96376747,0.00031075766,0.000113193964,0.00014238816,0.00031614097,0.0014810666,0.0024268695],"genre_scores_gemma":[0.6495016,0.00071103853,0.33931255,0.0003837473,0.00024842127,0.00025602608,0.0017182527,0.00041335906,0.0074549667],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99923885,0.00022982433,0.000038387945,0.00024263038,0.0001899938,0.00006037532],"domain_scores_gemma":[0.99877316,0.00057979976,0.00015412767,0.00013257282,0.00030019807,0.000060143913],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010060157,0.0011998668,0.00069382694,0.0013775057,0.0004532745,0.0011736165,0.0010813681,0.0009973518,0.002184385],"category_scores_gemma":[0.004131868,0.00037184404,0.00065445737,0.0010372343,0.0005992599,0.0020061191,0.0009537984,0.0014916877,0.00064683333],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004521191,0.00019585271,0.0018277877,0.00030071218,0.00014028019,0.0002353202,0.00027871082,0.31689313,0.030420844,0.013983491,0.008204768,0.6270669],"study_design_scores_gemma":[0.000009895452,0.000050834227,0.00038109426,0.000009472196,0.000021305086,0.000029614379,0.000019857824,0.9861432,0.005153752,0.0069782445,0.0011888872,0.0000138365385],"about_ca_topic_score_codex":0.0031968378,"about_ca_topic_score_gemma":0.005351826,"teacher_disagreement_score":0.0031968378,"about_ca_system_score_codex":0.0013375029,"about_ca_system_score_gemma":0.0005549936,"threshold_uncertainty_score":0.009704292},"labels":[],"label_agreement":null},{"id":"W3205031498","doi":"10.1145/3474085.3481545","title":"Personalized Multi-modal Video Retrieval on Mobile Devices","year":2021,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Centre for Social Innovation","funders":"","keywords":"Computer science; Information retrieval; Ranking (information retrieval); Context (archaeology); Mobile device; Matching (statistics); Process (computing); Modal; Face (sociological concept); World Wide Web","score_opus":0.02194809163755661,"score_gpt":0.31169704922897834,"score_spread":0.2897489575914217,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3205031498","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17176872,0.0017402964,0.789348,0.00055075757,0.00012546654,0.0005389699,0.0014978576,0.02335453,0.011075388],"genre_scores_gemma":[0.7781047,0.00058418623,0.20901802,0.0005184992,0.0001548045,0.00018909069,0.0018478204,0.00025760566,0.009325215],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989924,0.00022673803,0.000059890703,0.00029799776,0.000284834,0.00013805788],"domain_scores_gemma":[0.9989849,0.0002125188,0.00010191193,0.00040918993,0.00023365056,0.000057859655],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009937723,0.0006299246,0.0011549757,0.0011156709,0.00053366995,0.0009831777,0.0011522038,0.001065568,0.0024125965],"category_scores_gemma":[0.0030756737,0.00026555467,0.00051066396,0.0008904613,0.00027504066,0.0027079594,0.0012243488,0.0005822506,0.0018959732],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013264959,0.00040887913,0.004544566,0.00028642174,0.0001726542,0.0006625737,0.00040422403,0.036096413,0.15254472,0.0068334076,0.030223127,0.7664964],"study_design_scores_gemma":[0.0000867813,0.00044852932,0.008069348,0.000041540512,0.00012306345,0.0010259744,0.0005003229,0.8515932,0.1068824,0.012331864,0.018767003,0.00013002392],"about_ca_topic_score_codex":0.006376651,"about_ca_topic_score_gemma":0.009820661,"teacher_disagreement_score":0.006376651,"about_ca_system_score_codex":0.0008427608,"about_ca_system_score_gemma":0.0004658867,"threshold_uncertainty_score":0.0126791},"labels":[],"label_agreement":null},{"id":"W3211759539","doi":"","title":"Referring Transformer: A One-step Approach to Multi-task Visual Grounding","year":2021,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Transformer; Artificial intelligence; Segmentation; Encoder; Ground; Leverage (statistics); Margin (machine learning); Language model; Machine learning; Natural language processing; Engineering","score_opus":0.040190123109037815,"score_gpt":0.30488632255680165,"score_spread":0.26469619944776385,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3211759539","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0133652445,0.000483734,0.96736676,0.00033451497,0.00008770421,0.00016935557,0.00029141043,0.0146846045,0.0032166066],"genre_scores_gemma":[0.48040584,0.000517223,0.50195515,0.00083787646,0.00015988314,0.00032957076,0.0015554619,0.0012628424,0.012976174],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989687,0.00023325227,0.00004095257,0.00043141656,0.00017283853,0.00015290901],"domain_scores_gemma":[0.99903595,0.0003145874,0.00007312128,0.00033782638,0.0001571581,0.000081345985],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015640485,0.0018708577,0.0013556625,0.0009454204,0.00044586952,0.0015225398,0.0040681125,0.002296771,0.008222523],"category_scores_gemma":[0.0034333267,0.00069850753,0.0019169598,0.0010216439,0.0010449128,0.0038558105,0.0034307027,0.0026886389,0.003112856],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055833755,0.00039602502,0.00085379207,0.00031309342,0.0002120038,0.00039741845,0.0004753438,0.089838706,0.054974463,0.017464846,0.014071138,0.82044494],"study_design_scores_gemma":[0.00004165968,0.00024437887,0.000524283,0.000025054123,0.0000826632,0.00018660854,0.00009341588,0.93937445,0.024901103,0.028641567,0.005837368,0.000047360278],"about_ca_topic_score_codex":0.0050411522,"about_ca_topic_score_gemma":0.0067680073,"teacher_disagreement_score":0.008222523,"about_ca_system_score_codex":0.0009515886,"about_ca_system_score_gemma":0.001289959,"threshold_uncertainty_score":0.027507067},"labels":[],"label_agreement":null},{"id":"W3212471218","doi":"10.3390/app112110449","title":"Towards Contactless Learning Activities during Pandemics Using Autonomous Service Robots","year":2021,"lang":"en","type":"article","venue":"Applied Sciences","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Robot; Cheating; Personalization; Robotics; Artificial intelligence; Computer science; Service (business); Process (computing); Pandemic; Service-learning; Service robot; Human–computer interaction; Coronavirus disease 2019 (COVID-19); Psychology; Business; World Wide Web; Pedagogy; Marketing; Medicine","score_opus":0.03339591726116201,"score_gpt":0.2934661571553078,"score_spread":0.2600702398941458,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3212471218","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.47450116,0.00025363718,0.5148571,0.00032931136,0.000091450995,0.00020799952,0.00008886282,0.0054145604,0.00425584],"genre_scores_gemma":[0.8775095,0.00009154567,0.11974512,0.00010984674,0.000015267566,0.00007476123,0.000109068525,0.0000556716,0.0022892463],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9995908,0.00010559291,0.000018331037,0.00009184235,0.00013447266,0.000058882106],"domain_scores_gemma":[0.9992182,0.00016515535,0.00016822365,0.0001295904,0.00022246006,0.00009632049],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00049146486,0.0005778367,0.0004392871,0.0005987016,0.0004540118,0.000570945,0.0007491566,0.0008088309,0.0008950842],"category_scores_gemma":[0.0014876717,0.00020820287,0.00028098974,0.00027837817,0.0004842385,0.0007308449,0.00094100164,0.00045960004,0.0007045184],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004799449,0.00037378378,0.029349659,0.00034710902,0.00007102412,0.0009362122,0.0016966192,0.056736838,0.22579934,0.0019697614,0.004506477,0.67773324],"study_design_scores_gemma":[0.000073858515,0.0013903397,0.02666217,0.00006209056,0.00006691442,0.0010755829,0.0013448544,0.8470038,0.10595195,0.0026306126,0.013645362,0.00009251926],"about_ca_topic_score_codex":0.0016541453,"about_ca_topic_score_gemma":0.0014924478,"teacher_disagreement_score":0.0016541453,"about_ca_system_score_codex":0.00021735742,"about_ca_system_score_gemma":0.0005064727,"threshold_uncertainty_score":0.003289044},"labels":[],"label_agreement":null},{"id":"W3213111926","doi":"10.48550/arxiv.2012.02339","title":"Understanding Guided Image Captioning Performance across Domains","year":2020,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Closed captioning; Computer science; Readability; Transformer; Artificial intelligence; Natural language processing; Image (mathematics); Domain (mathematical analysis); Key (lock); Information retrieval; Programming language","score_opus":0.2005041772895836,"score_gpt":0.23192816123849722,"score_spread":0.0314239839489136,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3213111926","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.49501744,0.007687208,0.41614223,0.0018970398,0.0007545331,0.0012055704,0.00609647,0.04478002,0.026419451],"genre_scores_gemma":[0.79956377,0.0012506784,0.18149391,0.0006504825,0.00011578793,0.00030365802,0.0102114305,0.0012048166,0.0052054557],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99849176,0.0006512144,0.00007173213,0.0004984898,0.00016960084,0.00011715975],"domain_scores_gemma":[0.99050456,0.00620044,0.0003126616,0.0015944882,0.0010962667,0.00029155466],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00401172,0.0018277785,0.00065247196,0.0013585596,0.00049345894,0.0023051505,0.0018098648,0.002717609,0.0040146764],"category_scores_gemma":[0.02065178,0.0005104488,0.0010865213,0.00084359996,0.0008529149,0.0042927423,0.0016018952,0.0024853246,0.002921435],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012726551,0.0006022577,0.0065961955,0.0012931164,0.00039450196,0.00058107165,0.0015888698,0.29427215,0.044869952,0.0048203804,0.038089413,0.60561943],"study_design_scores_gemma":[0.00006355971,0.0003383823,0.0026366687,0.00012635368,0.000065722634,0.000265007,0.00037343366,0.9491901,0.031763073,0.008712155,0.0063974434,0.00006814171],"about_ca_topic_score_codex":0.008254639,"about_ca_topic_score_gemma":0.0066798506,"teacher_disagreement_score":0.008254639,"about_ca_system_score_codex":0.0016294795,"about_ca_system_score_gemma":0.0006916358,"threshold_uncertainty_score":0.021216273},"labels":[],"label_agreement":null},{"id":"W4200066496","doi":"10.1109/iros51168.2021.9636422","title":"Trajectory-Constrained Deep Latent Visual Attention for Improved Local Planning in Presence of Heterogeneous Terrain","year":2021,"lang":"en","type":"article","venue":"2021 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Artificial intelligence; Trajectory; Generalization; Terrain; Constraint (computer-aided design); Feature (linguistics); Task (project management); Computer vision; Collision avoidance; Visual space; Machine learning; Collision; Mathematics; Engineering; Psychology","score_opus":0.042992689432445066,"score_gpt":0.328001037531914,"score_spread":0.28500834809946896,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4200066496","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034747362,0.00018150597,0.96211404,0.00016363454,0.000033040415,0.000020213714,0.00006779046,0.0015554943,0.0011168877],"genre_scores_gemma":[0.90849787,0.00010914715,0.087676644,0.00016762996,0.000039809187,0.00005239041,0.00019467217,0.00019615449,0.0030656199],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998149,0.00003014235,0.000006375329,0.000062927225,0.00004261147,0.000043010354],"domain_scores_gemma":[0.9995171,0.00021899705,0.00006167777,0.000076586366,0.000072513285,0.000053051903],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00040832715,0.0007104967,0.00070017844,0.0003371662,0.00026402404,0.0005854791,0.0015065696,0.00072740164,0.0017341916],"category_scores_gemma":[0.0018478595,0.00041124783,0.0004992009,0.00039698975,0.0005263311,0.00091798895,0.0013248514,0.001264693,0.0003099733],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014661097,0.00007599305,0.0009104862,0.000047284033,0.000045234246,0.00007966396,0.00009020319,0.86336404,0.009212171,0.0041326224,0.0017784901,0.12011721],"study_design_scores_gemma":[0.0000047634935,0.000017450973,0.00009352801,0.0000019077838,0.0000036670149,0.000008927974,0.000002801641,0.99734735,0.000847659,0.0015479178,0.00012189397,0.000002163805],"about_ca_topic_score_codex":0.013959937,"about_ca_topic_score_gemma":0.015741313,"teacher_disagreement_score":0.013959937,"about_ca_system_score_codex":0.0009062807,"about_ca_system_score_gemma":0.0012491245,"threshold_uncertainty_score":0.027757406},"labels":[],"label_agreement":null},{"id":"W4200635486","doi":"10.1609/aaai.v36i3.20243","title":"MAGIC: Multimodal relAtional Graph adversarIal inferenCe for Diverse and Unpaired Text-Based Image Captioning","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"National Key Research and Development Program of China","keywords":"Closed captioning; Computer science; Artificial intelligence; Natural language processing; Inference; MAGIC (telescope); Sentence; Image (mathematics)","score_opus":0.05780263208315548,"score_gpt":0.305284387201188,"score_spread":0.24748175511803253,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4200635486","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014386917,0.00033779413,0.9795951,0.0004727744,0.00007907456,0.000099877,0.00032839822,0.0020084088,0.0026917227],"genre_scores_gemma":[0.7034452,0.0003299986,0.2819254,0.0011302847,0.00017209351,0.00034567635,0.0022164052,0.00062405196,0.009810957],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9994143,0.00021350745,0.000015977987,0.00019335256,0.00010484283,0.000057979385],"domain_scores_gemma":[0.998293,0.0011825267,0.00011766948,0.00019305477,0.00013903517,0.0000747336],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001236833,0.0013334157,0.0007893664,0.000525208,0.00043169552,0.0006855665,0.0022486825,0.0016379943,0.0050987247],"category_scores_gemma":[0.0048719277,0.0005833043,0.00086914294,0.00041181763,0.0011216273,0.0013583241,0.0017053605,0.0024897736,0.00097637187],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015121067,0.000075339856,0.00048871257,0.00011830975,0.00006377142,0.00020026583,0.00009904956,0.90134597,0.0047918144,0.014488218,0.0071359635,0.07104137],"study_design_scores_gemma":[0.000005011759,0.000013743991,0.000039523904,0.0000039442666,0.000004185653,0.000017501183,0.0000049342566,0.99389267,0.0008308754,0.004825522,0.0003582805,0.0000037480966],"about_ca_topic_score_codex":0.0031740675,"about_ca_topic_score_gemma":0.004654467,"teacher_disagreement_score":0.0050987247,"about_ca_system_score_codex":0.0010622441,"about_ca_system_score_gemma":0.00057441456,"threshold_uncertainty_score":0.017056942},"labels":[],"label_agreement":null},{"id":"W4206247875","doi":"10.1007/978-3-030-92659-5_11","title":"ScaleNet: An Unsupervised Representation Learning Method for Limited Information","year":2021,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Canadian Institute for Advanced Research","keywords":"Computer science; Artificial intelligence; Task (project management); Convolutional neural network; Representation (politics); Pattern recognition (psychology); Rotation (mathematics); Enhanced Data Rates for GSM Evolution; Scale (ratio); Machine learning; Artificial neural network; Deep learning; Feature learning","score_opus":0.025142420801092375,"score_gpt":0.31748589307844777,"score_spread":0.2923434722773554,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4206247875","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001661057,0.00016655223,0.99258655,0.00008218313,0.00006839012,0.000032141903,0.00026631373,0.00401548,0.0011212163],"genre_scores_gemma":[0.05738563,0.0004955646,0.92677176,0.00019442248,0.00014950753,0.0002748692,0.0025841405,0.0018597529,0.010284351],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994573,0.00012691478,0.000029523053,0.00013764034,0.00020237318,0.000046124933],"domain_scores_gemma":[0.99929214,0.0002511616,0.00003869648,0.00021519078,0.00016298289,0.00003981966],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010514726,0.0011826832,0.0011498852,0.0014254106,0.0006325455,0.0014007164,0.0020790454,0.0011988231,0.009155294],"category_scores_gemma":[0.002938026,0.0006861871,0.0010172233,0.0019286658,0.00073559885,0.0028813083,0.0019037043,0.0018838386,0.0044665285],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015600112,0.00008455039,0.0004086222,0.00017962675,0.00010763229,0.00010331271,0.00007976956,0.07201002,0.0087121455,0.037442677,0.03408994,0.8466256],"study_design_scores_gemma":[0.000019680141,0.000027148262,0.00017560185,0.000020143843,0.00002200207,0.0000667503,0.000020321439,0.9392544,0.0054461183,0.0424988,0.012430323,0.000018666986],"about_ca_topic_score_codex":0.004895855,"about_ca_topic_score_gemma":0.008578925,"teacher_disagreement_score":0.009155294,"about_ca_system_score_codex":0.00062358705,"about_ca_system_score_gemma":0.0010020913,"threshold_uncertainty_score":0.030627549},"labels":[],"label_agreement":null},{"id":"W4213017138","doi":"10.1145/3488560.3498507","title":"A GNN-based Multi-task Learning Framework for Personalized Video Search","year":2022,"lang":"en","type":"article","venue":"Proceedings of the Fifteenth ACM International Conference on Web Search and Data Mining","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Wilfrid Laurier University","funders":"","keywords":"Computer science; Task (project management); Human–computer interaction; Multimedia; Systems engineering; Engineering","score_opus":0.12830707427337426,"score_gpt":0.3845772540133766,"score_spread":0.2562701797400023,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4213017138","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.045623798,0.0017083599,0.94640285,0.0005752495,0.00013205981,0.00008369284,0.00037410556,0.0012850256,0.0038148435],"genre_scores_gemma":[0.84883,0.00072556705,0.14021793,0.0004808306,0.00015462682,0.00018105192,0.00076688937,0.0001123567,0.008530758],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997081,0.00007314131,0.000014964453,0.00009822621,0.000050236264,0.000055140343],"domain_scores_gemma":[0.99966407,0.00014023234,0.000035937825,0.00002862531,0.0001038501,0.000027243519],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006732659,0.0008031589,0.0009601598,0.00069061655,0.00031255832,0.0004717618,0.0015897835,0.0014049667,0.0017949116],"category_scores_gemma":[0.0017933454,0.0003969188,0.00070073846,0.00124641,0.00040998252,0.0012770188,0.0007206608,0.0012483122,0.00055394083],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013517393,0.00013012586,0.0012736518,0.0000958758,0.00006694506,0.00010307045,0.00007047534,0.8579677,0.0025356612,0.0059098983,0.0037329542,0.1279785],"study_design_scores_gemma":[0.0000033617102,0.000013296417,0.00008381304,0.0000018718981,0.000004471612,0.000008364587,0.0000029468642,0.998145,0.00009918498,0.0014849489,0.00015032648,0.0000023530417],"about_ca_topic_score_codex":0.027211271,"about_ca_topic_score_gemma":0.025866592,"teacher_disagreement_score":0.027211271,"about_ca_system_score_codex":0.0011530943,"about_ca_system_score_gemma":0.0009557578,"threshold_uncertainty_score":0.05410576},"labels":[],"label_agreement":null},{"id":"W4221149779","doi":"10.1109/icra46639.2022.9812195","title":"Generalizable task representation learning from human demonstration videos: a geometric approach","year":2022,"lang":"en","type":"article","venue":"2022 International Conference on Robotics and Automation (ICRA)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Robot; Artificial intelligence; Task (project management); Generalization; Representation (politics); Visual servoing; Computer vision; Task analysis; Categorical variable; Machine learning; Mathematics; Engineering","score_opus":0.05002559006014977,"score_gpt":0.30352058687195377,"score_spread":0.253494996811804,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4221149779","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021708023,0.0001767709,0.9765645,0.00020513215,0.000012850678,0.000076225326,0.000102860875,0.0006813595,0.00047227414],"genre_scores_gemma":[0.7050991,0.00038628938,0.2892034,0.0003521753,0.00007994399,0.00048205358,0.0010328521,0.00026016994,0.0031039375],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9989754,0.00031112923,0.000048351067,0.00042342357,0.00013898991,0.000102764345],"domain_scores_gemma":[0.9967212,0.0018638496,0.00038432938,0.00065835245,0.00024464575,0.00012754044],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016180997,0.0015158483,0.0013689597,0.0005505534,0.00027267067,0.0006303194,0.002358368,0.0021003874,0.0019982336],"category_scores_gemma":[0.0075233663,0.00077534706,0.0010318923,0.0006734299,0.0015326302,0.002212844,0.0019471815,0.0022903658,0.00035992105],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002236631,0.00015761431,0.00075665984,0.00024179433,0.00007109913,0.00017653008,0.00018498536,0.8081244,0.012663629,0.007122737,0.0017730935,0.16850376],"study_design_scores_gemma":[0.000025376352,0.00015027114,0.00033445473,0.000009732486,0.000007740133,0.0000476961,0.00002608446,0.9875195,0.0024212564,0.008966864,0.00047911855,0.000011866126],"about_ca_topic_score_codex":0.0048485287,"about_ca_topic_score_gemma":0.0037585797,"teacher_disagreement_score":0.0048485287,"about_ca_system_score_codex":0.0010642963,"about_ca_system_score_gemma":0.00097186735,"threshold_uncertainty_score":0.009640634},"labels":[],"label_agreement":null},{"id":"W4221155360","doi":"10.1109/cvpr52688.2022.00503","title":"MuKEA: Multimodal Knowledge Extraction and Accumulation for Knowledge-based Visual Question Answering","year":2022,"lang":"en","type":"article","venue":"2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":116,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"National Natural Science Foundation of China","keywords":"Computer science; Embedding; Knowledge extraction; Construct (python library); Domain knowledge; Question answering; Pipeline (software); Relation (database); Commonsense knowledge; Knowledge base; Artificial intelligence; Domain (mathematical analysis); Bridge (graph theory); Natural language processing; Data mining","score_opus":0.06767699411909972,"score_gpt":0.37926226941430574,"score_spread":0.311585275295206,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4221155360","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01224932,0.0015733882,0.91387904,0.0005336048,0.00017880255,0.00069760054,0.0050004916,0.060262565,0.0056252126],"genre_scores_gemma":[0.13435118,0.00067859975,0.83443063,0.00066899107,0.00009332745,0.0007664847,0.02143489,0.00085980754,0.006716046],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9986873,0.0003189505,0.00008118897,0.0005527477,0.00023520515,0.00012459148],"domain_scores_gemma":[0.9984035,0.00063021225,0.0000708567,0.0005234199,0.00026513738,0.000106860614],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001906699,0.0025184434,0.0012257245,0.0043481234,0.000892844,0.0020382951,0.0037581765,0.0026032138,0.019723216],"category_scores_gemma":[0.0063909423,0.0007797802,0.0024475895,0.001816134,0.00093344453,0.00587007,0.0059311176,0.0034959638,0.008593424],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037506653,0.00054553617,0.0019549832,0.0010322365,0.00032718095,0.00023745265,0.00072147883,0.017870879,0.017923854,0.0120062325,0.056407966,0.8905971],"study_design_scores_gemma":[0.0001520491,0.00032633953,0.003239434,0.00029496578,0.00029878633,0.00042940397,0.0008415524,0.80950075,0.028589597,0.078872845,0.07733202,0.00012220939],"about_ca_topic_score_codex":0.010847195,"about_ca_topic_score_gemma":0.019304,"teacher_disagreement_score":0.019723216,"about_ca_system_score_codex":0.0013892733,"about_ca_system_score_gemma":0.0016798002,"threshold_uncertainty_score":0.06598067},"labels":[],"label_agreement":null},{"id":"W4225683910","doi":"10.24963/ijcai.2022/762","title":"A Survey of Vision-Language Pre-Trained Models","year":2022,"lang":"en","type":"article","venue":"Proceedings of the Thirty-First International Joint Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":136,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Transformer; Pace; Artificial intelligence; Pointer (user interface); Deep learning; Mainstream; Focus (optics); Natural language processing; Human–computer interaction; Engineering","score_opus":0.06806887872685286,"score_gpt":0.3226548770219884,"score_spread":0.25458599829513556,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4225683910","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015976133,0.07578043,0.87865776,0.0028370707,0.00093394064,0.00019459843,0.0011031445,0.007855053,0.016661901],"genre_scores_gemma":[0.36185893,0.08828411,0.47875363,0.004141655,0.0016356803,0.00078138756,0.011870522,0.0024544918,0.0502197],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9994023,0.00015991599,0.00004790814,0.00020034004,0.00012535597,0.00006427346],"domain_scores_gemma":[0.99885356,0.00059085723,0.000037285256,0.00018518924,0.00028448633,0.000048612957],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011124833,0.0018073897,0.0011073633,0.0009548017,0.00037069106,0.0013613247,0.0028363978,0.0015130781,0.00615347],"category_scores_gemma":[0.0049510696,0.0007291263,0.0012160607,0.0012142834,0.0005826692,0.002801838,0.0011924254,0.0027793716,0.004400886],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016549169,0.00016573192,0.0015560144,0.00092827785,0.00016834197,0.000112410286,0.00014321771,0.07835431,0.0044256905,0.013907726,0.025212051,0.8748607],"study_design_scores_gemma":[0.000027026032,0.0003483755,0.0015831008,0.00069608586,0.00017172213,0.0003258375,0.00013056833,0.88694775,0.011167334,0.02512463,0.07339535,0.00008217558],"about_ca_topic_score_codex":0.009325035,"about_ca_topic_score_gemma":0.010376766,"teacher_disagreement_score":0.009325035,"about_ca_system_score_codex":0.0012704089,"about_ca_system_score_gemma":0.0017852503,"threshold_uncertainty_score":0.020585418},"labels":[],"label_agreement":null},{"id":"W4235356794","doi":"10.1109/cvpr.2019.00004","title":"Table of Contents","year":2019,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Microsoft Research Asia; Xidian University; Institute of Automation, Chinese Academy of Sciences; Council for Research in the Social Sciences, Columbia University; Beijing Institute of Technology; Chinese University of Hong Kong; College of Engineering, Michigan State University; Tongji University; Chinese Academy of Sciences; Connaught Fund; Pohang University of Science and Technology; Harbin Institute of Technology; Agency for Science, Technology and Research; Commonwealth Scientific and Industrial Research Organisation; De La Salle University; Eidgenössische Technische Hochschule Zürich; Tencent; National Laboratory of Pattern Recognition; Johns Hopkins University; Northwestern University; Microsoft Research","keywords":"Table (database); Computer science; Table of contents; Database; World Wide Web","score_opus":0.011581806221787398,"score_gpt":0.25684472083258575,"score_spread":0.24526291461079835,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4235356794","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0006591705,0.0033909692,0.0038186533,0.0025638032,0.007748501,0.0008389004,0.093336836,0.0029279664,0.88471514],"genre_scores_gemma":[0.0027168102,0.0030786183,0.0025379946,0.0020333568,0.0023756733,0.0006753488,0.06387949,0.0013080166,0.92139465],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99953794,0.00006105946,0.000026563377,0.000083790204,0.00024867782,0.000041963704],"domain_scores_gemma":[0.9969783,0.00064598484,0.00013130589,0.000296815,0.0015235709,0.00042406798],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.00059238414,0.0010972134,0.0010147623,0.0047573918,0.0013330095,0.0033840807,0.0014421949,0.0008653851,0.89209086],"category_scores_gemma":[0.006737726,0.00030315918,0.00051613746,0.0033092836,0.00030875753,0.0016468375,0.0016383334,0.0009223937,0.8341731],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000022709624,0.000035191642,0.0001440443,0.00021432612,0.0000040852324,0.000018704013,0.000009236421,0.00010603449,0.00017022248,0.00095441425,0.93986446,0.058456637],"study_design_scores_gemma":[0.000009292515,0.000018384428,0.00046077333,0.00023277439,0.0000058957753,0.000045004865,0.000025252979,0.000085591026,0.0001737466,0.0013455186,0.99759066,0.000007117117],"about_ca_topic_score_codex":0.0034663083,"about_ca_topic_score_gemma":0.0048203315,"teacher_disagreement_score":0.10790914,"about_ca_system_score_codex":0.0012636099,"about_ca_system_score_gemma":0.002280811,"threshold_uncertainty_score":0.15391928},"labels":[],"label_agreement":null},{"id":"W4249356701","doi":"10.1115/detc2018-85670","title":"Visual Similarity to Aid Alternative-Use Concept Generation for Retired Wind-Turbine Blades","year":2018,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reuse; Similarity (geometry); Turbine; Computer science; Scale (ratio); Turbine blade; Work (physics); Wind power; Artificial intelligence; Industrial engineering; Human–computer interaction; Engineering; Mechanical engineering; Image (mathematics)","score_opus":0.05234988231998867,"score_gpt":0.35103724004132,"score_spread":0.2986873577213313,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4249356701","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1411898,0.00043323627,0.84049886,0.00026257156,0.00011292933,0.00068556244,0.00046495078,0.005131857,0.011220359],"genre_scores_gemma":[0.33140558,0.00011835881,0.6644206,0.000080517035,0.000022367474,0.00020848829,0.0005776225,0.0002263921,0.002940079],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993647,0.0001674572,0.000033437056,0.00021690919,0.00017397523,0.00004353709],"domain_scores_gemma":[0.9978904,0.0012528123,0.00013840348,0.0003115986,0.00030816722,0.00009860208],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012279049,0.00090644695,0.00045742365,0.0018743051,0.00039009686,0.0011169694,0.001306816,0.0007630341,0.008256673],"category_scores_gemma":[0.005901147,0.0003355913,0.001046122,0.00067313755,0.0005405606,0.002197871,0.0015522704,0.00092496,0.0011545072],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047317526,0.00046984866,0.0032966537,0.00084046787,0.00010405419,0.00043464027,0.0027508845,0.013804623,0.051104844,0.021537986,0.005795629,0.89938724],"study_design_scores_gemma":[0.00033200072,0.0011846415,0.011815322,0.00031046342,0.0003168955,0.0020975217,0.0030753182,0.7017542,0.13036786,0.07769025,0.07083132,0.00022413916],"about_ca_topic_score_codex":0.0017028018,"about_ca_topic_score_gemma":0.0026596817,"teacher_disagreement_score":0.008256673,"about_ca_system_score_codex":0.0006714243,"about_ca_system_score_gemma":0.0008137794,"threshold_uncertainty_score":0.027621329},"labels":[],"label_agreement":null},{"id":"W4280596375","doi":"10.1111/cgf.14573","title":"Chart Question Answering: State of the Art and Future Directions","year":2022,"lang":"en","type":"article","venue":"Computer Graphics Forum","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Question answering; Computer science; Bar chart; Chart; Pie chart; Information retrieval; Task (project management); Domain (mathematical analysis); Data science; Artificial intelligence","score_opus":0.00627797374679749,"score_gpt":0.2314319068587537,"score_spread":0.22515393311195622,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4280596375","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017058238,0.78693116,0.12647405,0.037791364,0.0017137226,0.00056638883,0.001690981,0.005225856,0.022548243],"genre_scores_gemma":[0.20506899,0.5003086,0.25370172,0.01454688,0.0066240965,0.0010730999,0.010242631,0.0010201003,0.0074139573],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9804083,0.009113392,0.0011126747,0.0037469885,0.0043808366,0.0012378347],"domain_scores_gemma":[0.8907493,0.08396713,0.0019756923,0.0044482425,0.016548721,0.0023110234],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.023502862,0.0018451911,0.0023844978,0.007835067,0.0015240881,0.011402482,0.008174886,0.0062926896,0.015483828],"category_scores_gemma":[0.051862534,0.00092140946,0.0021548234,0.009739172,0.0040401164,0.025824776,0.0043315575,0.005232886,0.0059806677],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000358842,0.0005439787,0.0035478184,0.01035837,0.00014676903,0.00009469066,0.0016489786,0.0043760375,0.0018273152,0.033464108,0.048110947,0.8955221],"study_design_scores_gemma":[0.000192555,0.0008442239,0.010678084,0.01521056,0.00051285553,0.0008849146,0.010342222,0.13057864,0.005198038,0.2288383,0.5962756,0.00044404733],"about_ca_topic_score_codex":0.009836104,"about_ca_topic_score_gemma":0.004464949,"teacher_disagreement_score":0.023502862,"about_ca_system_score_codex":0.0031142076,"about_ca_system_score_gemma":0.005556124,"threshold_uncertainty_score":0.12429649},"labels":[],"label_agreement":null},{"id":"W4281760878","doi":"10.3390/electronics11111785","title":"Deep Learning-Based Context-Aware Video Content Analysis on IoT Devices","year":2022,"lang":"en","type":"article","venue":"Electronics","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Thunder Bay Regional Research Institute; Lakehead University","funders":"","keywords":"Computer science; Closed captioning; Artificial intelligence; Deep learning; Inference; Transformer; Sentence; Language model; Hyperparameter; Machine learning; Context (archaeology); Natural language processing; Image (mathematics)","score_opus":0.017826067059981115,"score_gpt":0.2683236016153511,"score_spread":0.25049753455537,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4281760878","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13122763,0.0017825163,0.85202354,0.00046074225,0.0002444236,0.00017830152,0.0015669187,0.0049085533,0.00760735],"genre_scores_gemma":[0.78237927,0.0011207289,0.20928481,0.00025173003,0.00012992401,0.00012474631,0.0024580741,0.00017126746,0.0040794504],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998647,0.000019837013,0.0000073816377,0.000043451997,0.000038640705,0.000025997124],"domain_scores_gemma":[0.9998584,0.000041514944,0.000017646391,0.000017310294,0.000053957134,0.000011145371],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001753559,0.0006147909,0.00039070926,0.0006950013,0.00023330486,0.0004330984,0.00051415374,0.00035010884,0.0014758495],"category_scores_gemma":[0.00087333226,0.00019199302,0.00041404093,0.00062360964,0.00018013017,0.00092644175,0.0005193962,0.0005140023,0.0006074422],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005992078,0.00026066272,0.004184831,0.00039313958,0.000104277475,0.0006993028,0.0001849811,0.1650023,0.10694187,0.005936321,0.014344134,0.70134896],"study_design_scores_gemma":[0.0000067691194,0.000041908046,0.001639906,0.000021635871,0.000020621343,0.00006650362,0.000058675923,0.9756328,0.017569546,0.0028435912,0.0020875994,0.000010289919],"about_ca_topic_score_codex":0.0045188656,"about_ca_topic_score_gemma":0.006694109,"teacher_disagreement_score":0.0045188656,"about_ca_system_score_codex":0.00053154485,"about_ca_system_score_gemma":0.00041946615,"threshold_uncertainty_score":0.008985102},"labels":[],"label_agreement":null},{"id":"W4283805152","doi":"10.1609/aaai.v36i2.20123","title":"Improving Zero-Shot Phrase Grounding via Reasoning on External Knowledge and Spatial Relations","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Phrase; Computer science; Noun phrase; Artificial intelligence; Ground; Visual reasoning; Natural language processing; Modal; Noun; Engineering","score_opus":0.04921381018470201,"score_gpt":0.30577594182359497,"score_spread":0.25656213163889297,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4283805152","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.067057796,0.001368768,0.89387447,0.0006804366,0.00017099382,0.00030815921,0.0030719943,0.026425123,0.007042196],"genre_scores_gemma":[0.44418505,0.0006731374,0.52865285,0.00087741023,0.000113245376,0.00014827917,0.016525768,0.0010903468,0.0077339406],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99882096,0.00013875798,0.0000546286,0.00050420745,0.00033908043,0.0001422554],"domain_scores_gemma":[0.9982109,0.0008473943,0.00012345644,0.0004909037,0.00024664047,0.00008074033],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010057268,0.0022043379,0.0014525454,0.0022830311,0.0009323304,0.0016887682,0.0038174684,0.0020272126,0.0062722885],"category_scores_gemma":[0.0047575925,0.00060298917,0.0014856658,0.0017756214,0.0015534282,0.0074845646,0.003992333,0.0025760024,0.002341981],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042007735,0.0005517828,0.003788483,0.0005608319,0.0001653922,0.0006319305,0.00055309356,0.059228346,0.030855415,0.019293176,0.031691615,0.85225976],"study_design_scores_gemma":[0.000100460085,0.00026673308,0.0016637902,0.00009013656,0.00015182696,0.00039744558,0.00056745956,0.86852133,0.035718493,0.08109743,0.011368083,0.000056813995],"about_ca_topic_score_codex":0.0143044265,"about_ca_topic_score_gemma":0.029964017,"teacher_disagreement_score":0.0143044265,"about_ca_system_score_codex":0.0012570496,"about_ca_system_score_gemma":0.0019255604,"threshold_uncertainty_score":0.028442323},"labels":[],"label_agreement":null},{"id":"W4287891193","doi":"10.18653/v1/2022.wordplay-1.4","title":"A Sequence Modelling Approach to Question Answering in Text-Based Games","year":2022,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institut National de la Recherche Scientifique","funders":"","keywords":"Question answering; Computer science; Transformer; Reinforcement learning; Artificial intelligence; Machine learning; Benchmark (surveying); Language model; Task (project management); Natural language processing","score_opus":0.03687691377713465,"score_gpt":0.2905471630438144,"score_spread":0.25367024926667975,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4287891193","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006837487,0.000113913076,0.98934877,0.00033740225,0.00003152278,0.0001026865,0.00010209827,0.0003575327,0.0027685885],"genre_scores_gemma":[0.6581198,0.00036666627,0.32570472,0.00037634347,0.0001184969,0.0005742705,0.00048752374,0.00015751617,0.014094756],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99918646,0.0004063124,0.00003946169,0.000184401,0.0001230357,0.00006039473],"domain_scores_gemma":[0.998572,0.0010382731,0.00008480954,0.00008920351,0.00012604844,0.000089647736],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011293364,0.000785337,0.0006584872,0.00081253034,0.0004587021,0.00094569975,0.0020686463,0.0014626858,0.0060466733],"category_scores_gemma":[0.004723603,0.00046639013,0.0011902727,0.0005503015,0.0011895308,0.001850942,0.0011439616,0.0018155513,0.00088810996],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015633281,0.00020153732,0.0008265752,0.00018337741,0.00006537414,0.00018102404,0.00059952994,0.7379671,0.005339856,0.18917745,0.0022101046,0.06309175],"study_design_scores_gemma":[0.000009281012,0.00003075662,0.000059221074,0.0000045388038,0.000004754802,0.000014622808,0.000011997862,0.9648967,0.00037873234,0.033782423,0.00080182793,0.00000513047],"about_ca_topic_score_codex":0.010230528,"about_ca_topic_score_gemma":0.009912958,"teacher_disagreement_score":0.010230528,"about_ca_system_score_codex":0.0014801098,"about_ca_system_score_gemma":0.0012395208,"threshold_uncertainty_score":0.020341992},"labels":[],"label_agreement":null},{"id":"W4292409471","doi":"10.3390/sym14081715","title":"Utilizing Language Models to Expand Vision-Based Commonsense Knowledge Graphs","year":2022,"lang":"en","type":"article","venue":"Symmetry","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Commonsense knowledge; Computer science; Commonsense reasoning; Language model; Artificial intelligence; Transformer; Embedding; Natural language processing; Question answering; Knowledge base; Natural language understanding; Knowledge graph; Natural language","score_opus":0.02364263878367856,"score_gpt":0.31271794585807844,"score_spread":0.2890753070743999,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4292409471","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.067591056,0.00027961595,0.92333233,0.0004691268,0.00006172225,0.00009063493,0.00042090652,0.0030915812,0.0046631317],"genre_scores_gemma":[0.71351177,0.00030840357,0.27816325,0.00039353926,0.00003859656,0.0001670146,0.0015726258,0.0004901007,0.005354653],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996425,0.00007466724,0.000017573027,0.0001427542,0.00008585234,0.00003680661],"domain_scores_gemma":[0.9988456,0.0006787978,0.000087114364,0.00017582571,0.00015456484,0.000058088594],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00047990802,0.0005926755,0.00037640386,0.0008547173,0.00033543105,0.000820878,0.0011179029,0.00064806995,0.0025499954],"category_scores_gemma":[0.0032464573,0.00034731964,0.0011336036,0.0004901666,0.0009358694,0.0032820874,0.0014495538,0.001402101,0.0007961046],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002526856,0.0003200754,0.0024432025,0.00042841386,0.00009607452,0.0008939327,0.0014366546,0.40989727,0.05150434,0.12500699,0.008057173,0.3996632],"study_design_scores_gemma":[0.000014942993,0.000040825522,0.0002444246,0.000023615226,0.000020254303,0.00009730534,0.000100857076,0.91539216,0.008273616,0.071424164,0.0043462627,0.000021515118],"about_ca_topic_score_codex":0.0030652054,"about_ca_topic_score_gemma":0.0058828783,"teacher_disagreement_score":0.0030652054,"about_ca_system_score_codex":0.00067674,"about_ca_system_score_gemma":0.00069502014,"threshold_uncertainty_score":0.008530617},"labels":[],"label_agreement":null},{"id":"W4292937774","doi":"10.17760/d20429166","title":"Improving instruction generation for vision-language navigation by reward designing","year":2021,"lang":"en","type":"dissertation","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Science North","funders":"","keywords":"Computer science; Generator (circuit theory); Function (biology); Artificial intelligence; Language model; Human–computer interaction","score_opus":0.009269301080167717,"score_gpt":0.30295176087001363,"score_spread":0.2936824597898459,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4292937774","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05732292,0.00026355186,0.9309371,0.0002878587,0.00012624364,0.00014819519,0.00018368203,0.0082512,0.0024792997],"genre_scores_gemma":[0.62782997,0.00014060449,0.36354604,0.00028295934,0.000051867763,0.00031856901,0.00082679465,0.0007187346,0.006284493],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994124,0.00017150784,0.000031725696,0.00019456599,0.00011798953,0.00007194118],"domain_scores_gemma":[0.9985801,0.0006840788,0.00009073065,0.00025113442,0.00031456028,0.00007953541],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012480826,0.0009751623,0.0007282003,0.00057897647,0.00038123233,0.00070714246,0.0014658894,0.0010230815,0.005852962],"category_scores_gemma":[0.005077878,0.00031092396,0.0006068838,0.00050153316,0.00058332045,0.0016781866,0.001405911,0.0014276623,0.0016689022],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004017304,0.00035669302,0.0034820586,0.00016107257,0.00005100264,0.00013878559,0.00016646332,0.26559952,0.024229359,0.0084689725,0.006653455,0.6902909],"study_design_scores_gemma":[0.000024220759,0.00011343172,0.0002984048,0.000008483043,0.000013648755,0.000038724673,0.000013016209,0.98159975,0.012869751,0.0036812655,0.0013277205,0.000011472117],"about_ca_topic_score_codex":0.0032834355,"about_ca_topic_score_gemma":0.004742435,"teacher_disagreement_score":0.005852962,"about_ca_system_score_codex":0.000906828,"about_ca_system_score_gemma":0.0013877183,"threshold_uncertainty_score":0.019580126},"labels":[],"label_agreement":null},{"id":"W4293868279","doi":"10.1109/crv55824.2022.00038","title":"3DVQA: Visual Question Answering for 3D Environments","year":2022,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada; Western Canada Research Grid; Compute Canada","keywords":"Question answering; Computer science; Task (project management); Artificial intelligence; Domain (mathematical analysis); Modality (human–computer interaction); Relation (database); Point (geometry); Baseline (sea); Natural language processing; Information retrieval; Computer vision; Data mining","score_opus":0.009262272345025406,"score_gpt":0.28865129407987955,"score_spread":0.27938902173485414,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4293868279","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08280785,0.015897889,0.2805194,0.004254646,0.0013639265,0.004747027,0.37406573,0.19318348,0.043159977],"genre_scores_gemma":[0.13455316,0.0018228849,0.25845507,0.002252154,0.0002187999,0.0022597983,0.5915077,0.0022085092,0.0067219785],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99576366,0.0011861194,0.00032671972,0.001252186,0.0011840399,0.00028720425],"domain_scores_gemma":[0.9945655,0.0023954466,0.00026682633,0.0017306489,0.0007045093,0.00033700204],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00256378,0.003990883,0.0020869125,0.004225203,0.0014914477,0.0030976892,0.0063256836,0.006019084,0.018261418],"category_scores_gemma":[0.014811609,0.00086928584,0.003245186,0.0026341283,0.0013921626,0.0073408564,0.007800746,0.003621586,0.010828854],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001061761,0.0007180875,0.0058176047,0.0051397304,0.00045512296,0.00074637705,0.0009856868,0.02836282,0.014195439,0.015222259,0.59538084,0.33191413],"study_design_scores_gemma":[0.00075315806,0.00075377274,0.012756941,0.0009508587,0.00019526632,0.0021470406,0.0016658322,0.37492797,0.020166114,0.07095529,0.514486,0.0002416991],"about_ca_topic_score_codex":0.028235858,"about_ca_topic_score_gemma":0.036497783,"teacher_disagreement_score":0.028235858,"about_ca_system_score_codex":0.002273995,"about_ca_system_score_gemma":0.0021628994,"threshold_uncertainty_score":0.06109047},"labels":[],"label_agreement":null},{"id":"W4293868331","doi":"10.1109/crv55824.2022.00015","title":"Semi-supervised Grounding Alignment for Multi-modal Feature Learning","year":2022,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; Canadian Institute for Advanced Research; University of British Columbia","funders":"","keywords":"Computer science; Artificial intelligence; Phrase; Margin (machine learning); Leverage (statistics); Machine learning; Ground; Sentence; Feature (linguistics); Transformer; Modal; Classifier (UML); Feature learning; Natural language processing; Question answering; Supervised learning; Feature extraction; Pattern recognition (psychology); Artificial neural network; Engineering","score_opus":0.031631322508959094,"score_gpt":0.30216253552996014,"score_spread":0.270531213021001,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4293868331","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010943403,0.00015361747,0.9832138,0.00014576869,0.000023527286,0.000066605266,0.00026441005,0.0044142366,0.0007746669],"genre_scores_gemma":[0.5182625,0.00018662296,0.47060588,0.00056843366,0.00014089713,0.00048119173,0.0043947296,0.00076006417,0.0045997156],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.998257,0.0004911006,0.00007080793,0.0007309109,0.0003059199,0.00014422367],"domain_scores_gemma":[0.9975636,0.0009129492,0.00027927294,0.0008015276,0.00034461523,0.000098135235],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020419864,0.0017702356,0.0012697938,0.00114315,0.0005787144,0.0011286776,0.0038160726,0.0021708235,0.004801504],"category_scores_gemma":[0.0055921543,0.00058933685,0.001113985,0.0014235075,0.0015571074,0.004359047,0.0029751528,0.0031408481,0.0023209588],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037079692,0.00042904503,0.0014878812,0.00022884649,0.00012076673,0.00013405242,0.00022636207,0.23916817,0.024397328,0.018161817,0.0156143475,0.6996606],"study_design_scores_gemma":[0.000013347378,0.00008072792,0.00021875443,0.000009131072,0.0000080574355,0.00003754491,0.00002205325,0.97562206,0.004936863,0.017952055,0.0010884913,0.000010930402],"about_ca_topic_score_codex":0.0022933676,"about_ca_topic_score_gemma":0.0037616396,"teacher_disagreement_score":0.004801504,"about_ca_system_score_codex":0.0013864553,"about_ca_system_score_gemma":0.00097162975,"threshold_uncertainty_score":0.016062677},"labels":[],"label_agreement":null},{"id":"W4297688052","doi":"10.1093/jcde/qwac084","title":"TransNav: spatial sequential transformer network for visual navigation","year":2022,"lang":"en","type":"article","venue":"Journal of Computational Design and Engineering","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Key Research and Development Program of China; Ministry of Natural Resources","keywords":"Computer science; Reinforcement learning; Artificial intelligence; Inference; Transformer; Machine learning; Engineering","score_opus":0.012894214952839501,"score_gpt":0.2588550867853863,"score_spread":0.24596087183254683,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4297688052","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06758702,0.0016118545,0.9036824,0.0006712142,0.00033373025,0.00012871971,0.0021297461,0.014339802,0.009515453],"genre_scores_gemma":[0.83768475,0.0007028612,0.1441861,0.00037942524,0.00006721703,0.00019594439,0.004014095,0.00044706278,0.012322529],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998435,0.000022777605,0.000006709375,0.00006133805,0.00003465881,0.00003106262],"domain_scores_gemma":[0.9997619,0.000074943346,0.000022012468,0.00004147042,0.0000697358,0.000029905992],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00031952633,0.0010737742,0.00069834816,0.00052760914,0.000309045,0.0005803832,0.0020619535,0.0008567619,0.0048980047],"category_scores_gemma":[0.0012162745,0.00041254162,0.00067883387,0.0005375145,0.00049491116,0.0013308948,0.0009793605,0.0014111338,0.0010059376],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037918752,0.00019329142,0.0017968566,0.00017020044,0.00010082101,0.0001726438,0.0000688619,0.68733644,0.00675053,0.013812041,0.01700461,0.2722146],"study_design_scores_gemma":[0.0000105011195,0.000023395585,0.00007843586,0.0000053100484,0.0000074920763,0.000013450056,0.000004390294,0.9937831,0.0009698616,0.004350415,0.0007494688,0.0000041320477],"about_ca_topic_score_codex":0.024596673,"about_ca_topic_score_gemma":0.027236653,"teacher_disagreement_score":0.024596673,"about_ca_system_score_codex":0.0010893107,"about_ca_system_score_gemma":0.0014118709,"threshold_uncertainty_score":0.048906982},"labels":[],"label_agreement":null},{"id":"W4306882122","doi":"10.48550/arxiv.2101.07241","title":"Learning by Watching: Physical Imitation of Manipulation Skills from Human Videos","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Canadian Institute for Advanced Research","keywords":"Computer science; Artificial intelligence; Reinforcement learning; Task (project management); Salient; Imitation; Robot; Representation (politics); Unsupervised learning; Machine learning; Deep learning; Human–computer interaction","score_opus":0.0409648179073554,"score_gpt":0.21768011579033145,"score_spread":0.17671529788297605,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4306882122","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030738618,0.00019425948,0.9659864,0.00015322439,0.000023378161,0.000070500995,0.00010249727,0.0015511803,0.0011800369],"genre_scores_gemma":[0.7644046,0.00026615354,0.23118508,0.00017806086,0.000045446694,0.00022261831,0.00044509122,0.00021746386,0.00303549],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9996145,0.000110626635,0.000014752468,0.00014752898,0.00007176856,0.000040825234],"domain_scores_gemma":[0.99880743,0.0006955003,0.0001541331,0.0001984362,0.000080940204,0.00006348937],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00078638503,0.0007564764,0.00064745994,0.0004240048,0.00023435827,0.0005577214,0.0015277715,0.0009485661,0.0014824718],"category_scores_gemma":[0.004515306,0.00046273565,0.00052601134,0.00032008215,0.0010454674,0.0012898798,0.00093528547,0.0011738507,0.0003101779],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023750463,0.00021738478,0.0027526529,0.00022006631,0.00008916167,0.00028345292,0.00024174905,0.5594494,0.024768582,0.014832833,0.003308125,0.39359906],"study_design_scores_gemma":[0.000011543725,0.00007239731,0.0003724529,0.000008563479,0.0000061888245,0.000044453678,0.000014412809,0.9877175,0.004092167,0.0070575983,0.00059314154,0.000009609144],"about_ca_topic_score_codex":0.0037056576,"about_ca_topic_score_gemma":0.0038868336,"teacher_disagreement_score":0.0037056576,"about_ca_system_score_codex":0.0006297118,"about_ca_system_score_gemma":0.0007713568,"threshold_uncertainty_score":0.0073681474},"labels":[],"label_agreement":null},{"id":"W4307475428","doi":"10.1145/3526113.3545621","title":"Opal: Multimodal Image Generation for News Illustration","year":2022,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":96,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"USable; Computer science; Pipeline (software); Image (mathematics); Tone (literature); Multimedia; Artificial intelligence; Human–computer interaction; Linguistics; Programming language","score_opus":0.03235914812416576,"score_gpt":0.29676236432622527,"score_spread":0.26440321620205953,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4307475428","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.047219217,0.0010812657,0.6951739,0.00077637006,0.00042742342,0.00080259296,0.0044080396,0.20569845,0.04441281],"genre_scores_gemma":[0.31756246,0.00088144495,0.615178,0.00080963847,0.0001848374,0.0011119414,0.009267103,0.01342721,0.04157737],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9996618,0.00010402204,0.00001896166,0.00007059431,0.000112586786,0.000032042648],"domain_scores_gemma":[0.9987778,0.00078508264,0.000048488473,0.0001873678,0.00011372769,0.000087570894],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00066009385,0.0010661498,0.00034012183,0.0009292765,0.0003584223,0.0013768971,0.0012213145,0.00093037327,0.04092586],"category_scores_gemma":[0.0041767447,0.00030879286,0.0007123771,0.00032237658,0.00047063886,0.0019408857,0.0026649083,0.00082027283,0.0075009232],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015086674,0.00043792088,0.002114498,0.0014894343,0.00014126251,0.0016386373,0.0021892348,0.0074486495,0.06867528,0.01598488,0.17302483,0.7253468],"study_design_scores_gemma":[0.000660887,0.0007773051,0.0051153037,0.00029717176,0.00015871767,0.0033832442,0.0010925084,0.3153363,0.14084667,0.036379416,0.49568164,0.00027077558],"about_ca_topic_score_codex":0.00055091584,"about_ca_topic_score_gemma":0.0008944009,"teacher_disagreement_score":0.04092586,"about_ca_system_score_codex":0.00029763876,"about_ca_system_score_gemma":0.00023192093,"threshold_uncertainty_score":0.13691062},"labels":[],"label_agreement":null},{"id":"W4308222681","doi":"10.1145/3536221.3556597","title":"A Framework for Video-Text Retrieval with Noisy Supervision","year":2022,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Closed captioning; Artificial intelligence; Modal; Relevance (law); Context (archaeology); Task (project management); Natural language processing; Domain (mathematical analysis); Machine learning; Information retrieval; Image (mathematics)","score_opus":0.01734851815095571,"score_gpt":0.2900835527978527,"score_spread":0.27273503464689697,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4308222681","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000992337,0.00023415848,0.9969073,0.00018256276,0.00002961372,0.00006754017,0.00020463733,0.000845797,0.0005360263],"genre_scores_gemma":[0.109013624,0.0007141618,0.8787915,0.00055472873,0.00033943585,0.0009329414,0.0028198115,0.0004564493,0.006377316],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981664,0.000559186,0.00011080686,0.00062037446,0.0004269971,0.000116347626],"domain_scores_gemma":[0.9981021,0.00079389574,0.00020142361,0.0003817605,0.00041159557,0.000109246634],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033506893,0.0012334613,0.0014714026,0.0018589569,0.00084619963,0.002007094,0.0038720733,0.0021374656,0.0034637626],"category_scores_gemma":[0.0077026053,0.00073621556,0.0015731968,0.00210449,0.001320108,0.0031202768,0.0028782263,0.002313948,0.002064795],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038742536,0.00032360785,0.00092087657,0.0005364422,0.0002361735,0.000439223,0.00045722412,0.45183292,0.0129019795,0.16660528,0.031198625,0.33416024],"study_design_scores_gemma":[0.000020413065,0.000044983943,0.00012001177,0.000016939865,0.000018088422,0.000072552895,0.000024759524,0.9545362,0.0012622812,0.038959157,0.004903167,0.000021371785],"about_ca_topic_score_codex":0.012482449,"about_ca_topic_score_gemma":0.014274224,"teacher_disagreement_score":0.012482449,"about_ca_system_score_codex":0.0018808766,"about_ca_system_score_gemma":0.0021790254,"threshold_uncertainty_score":0.024819613},"labels":[],"label_agreement":null},{"id":"W4310921857","doi":"10.48550/arxiv.2212.03338","title":"Framework-agnostic Semantically-aware Global Reasoning for Segmentation","year":2022,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; Canadian Institute for Advanced Research; University of British Columbia","funders":"","keywords":"Computer science; Segmentation; Artificial intelligence; Pixel; Representation (politics); Semantics (computer science); Object (grammar); Transformer; Natural language processing; Pattern recognition (psychology); Machine learning","score_opus":0.055302229931168075,"score_gpt":0.25294454173164577,"score_spread":0.1976423118004777,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4310921857","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013673754,0.00028793496,0.9746343,0.00017308512,0.000036323054,0.00007622637,0.00038343816,0.009040106,0.0016948873],"genre_scores_gemma":[0.39544353,0.0003707138,0.59655297,0.00035450587,0.000049605424,0.00010294109,0.0026439768,0.00087758154,0.0036041872],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99935347,0.00009810133,0.00003222107,0.00025482656,0.00017014705,0.00009131658],"domain_scores_gemma":[0.99938536,0.00015436044,0.000052562744,0.00025641854,0.00010413221,0.000047199377],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012056539,0.0012418684,0.00092992873,0.0009751042,0.0004344209,0.0015166302,0.0028127814,0.0012636676,0.004018007],"category_scores_gemma":[0.0022990578,0.0005576759,0.0017373196,0.00088442175,0.0011740045,0.0040096077,0.002270042,0.0022378615,0.0015740372],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048508382,0.00029365375,0.0020693168,0.00039888622,0.00022449253,0.00030939296,0.00045951916,0.2527364,0.072302796,0.051304642,0.012348456,0.6070674],"study_design_scores_gemma":[0.000020112175,0.000060201655,0.00035768535,0.000018317338,0.00005023493,0.00008374576,0.000050638344,0.93205184,0.018894078,0.045079406,0.0033164748,0.00001732664],"about_ca_topic_score_codex":0.006585846,"about_ca_topic_score_gemma":0.013107386,"teacher_disagreement_score":0.006585846,"about_ca_system_score_codex":0.0015364467,"about_ca_system_score_gemma":0.0016965638,"threshold_uncertainty_score":0.013441563},"labels":[],"label_agreement":null},{"id":"W4312261477","doi":"10.1109/cvpr52688.2022.00517","title":"Winoground: Probing Vision and Language Models for Visio-Linguistic Compositionality","year":2022,"lang":"en","type":"article","venue":"2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":185,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Principle of compositionality; Computer science; Set (abstract data type); Task (project management); Artificial intelligence; Natural language processing; Language model; Field (mathematics); State (computer science); Programming language","score_opus":0.042859049951476526,"score_gpt":0.31686812750494875,"score_spread":0.2740090775534722,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312261477","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27569255,0.010249175,0.12721244,0.0052352217,0.002466207,0.0028597575,0.42082247,0.097468965,0.057993304],"genre_scores_gemma":[0.22134791,0.00076268625,0.13290899,0.0020262254,0.00029044523,0.0013005288,0.6286279,0.0037564924,0.008978853],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9953696,0.0014104119,0.00028432664,0.0015510559,0.00094834267,0.00043630856],"domain_scores_gemma":[0.99329656,0.0026809464,0.0003788262,0.002401344,0.0008720488,0.0003702777],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037362683,0.004465636,0.0014097717,0.004003144,0.0021085627,0.0044037593,0.0056425403,0.0043217344,0.012365078],"category_scores_gemma":[0.014982793,0.0008695145,0.0027179294,0.0023594687,0.0020709618,0.006637028,0.0051171896,0.004234019,0.012185634],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020568932,0.0012852163,0.014686835,0.003215835,0.0006189026,0.0008452618,0.0007536715,0.026865242,0.014960308,0.012126341,0.70737153,0.21521394],"study_design_scores_gemma":[0.0013376728,0.0013449065,0.022031426,0.0010430064,0.00042108804,0.0040686857,0.002689083,0.47103786,0.052891206,0.04023012,0.40246242,0.00044246676],"about_ca_topic_score_codex":0.023821369,"about_ca_topic_score_gemma":0.04241925,"teacher_disagreement_score":0.023821369,"about_ca_system_score_codex":0.0028138063,"about_ca_system_score_gemma":0.0018338645,"threshold_uncertainty_score":0.047365367},"labels":[],"label_agreement":null},{"id":"W4312330522","doi":"10.1007/978-3-031-20044-1_16","title":"Open-World Semantic Segmentation via Contrasting and Clustering Vision-Language Embedding","year":2022,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada)","funders":"","keywords":"Computer science; Segmentation; Artificial intelligence; Pipeline (software); Cluster analysis; Image segmentation; Encoder; Embedding; Semantics (computer science); Object (grammar); Segmentation-based object categorization; Natural language processing; Pattern recognition (psychology); Scale-space segmentation; Computer vision","score_opus":0.013622829848246207,"score_gpt":0.30939730922828884,"score_spread":0.29577447938004264,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312330522","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013137918,0.00026453612,0.97885203,0.00013398484,0.00007639943,0.000067687506,0.00027361617,0.0033530553,0.0038407824],"genre_scores_gemma":[0.25056785,0.0005294568,0.7360241,0.00019308201,0.00009698131,0.00013974901,0.0030621416,0.0015276666,0.007858878],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992219,0.00010839232,0.00003897307,0.00034898106,0.00014960679,0.00013210851],"domain_scores_gemma":[0.99941206,0.00013215154,0.000046446923,0.00018703354,0.00016761085,0.00005470051],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000691706,0.0013529165,0.0012052062,0.0022670547,0.0008713321,0.0032750242,0.002303997,0.0017544532,0.0054719904],"category_scores_gemma":[0.0017400559,0.0007438432,0.0017328216,0.0028759788,0.0013000049,0.004079548,0.003314484,0.0018971694,0.00487588],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036991513,0.00021849995,0.00078144664,0.00026229848,0.000118692966,0.00026799552,0.0004323989,0.0525857,0.06055855,0.051073693,0.010136781,0.823194],"study_design_scores_gemma":[0.000018045692,0.00008401264,0.0006774464,0.000043126725,0.000055096305,0.00023857498,0.00027390607,0.88433117,0.03514801,0.069857985,0.009223359,0.00004931793],"about_ca_topic_score_codex":0.0050713657,"about_ca_topic_score_gemma":0.007852389,"teacher_disagreement_score":0.0054719904,"about_ca_system_score_codex":0.0010191605,"about_ca_system_score_gemma":0.0014220793,"threshold_uncertainty_score":0.01830566},"labels":[],"label_agreement":null},{"id":"W4312372711","doi":"10.1109/cvpr52688.2022.00495","title":"X-Pool: Cross-Modal Language-Video Attention for Text-Video Retrieval","year":2022,"lang":"en","type":"article","venue":"2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":204,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Toronto","funders":"","keywords":"Computer science; Focus (optics); Information retrieval; Representation (politics); ENCODE; Pooling; Recall; Natural language processing; Artificial intelligence; Similarity (geometry); Text retrieval; Code (set theory); Function (biology); Image (mathematics); Linguistics","score_opus":0.03439024797878233,"score_gpt":0.32644121003245424,"score_spread":0.2920509620536719,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312372711","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043586787,0.0051997746,0.91489786,0.0005577985,0.0003960676,0.0006978968,0.0016457632,0.02789606,0.005121961],"genre_scores_gemma":[0.47154903,0.0018662845,0.4915252,0.0022871254,0.0006040695,0.0010098834,0.0070630233,0.0011657956,0.02292954],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999424,0.000090424495,0.000029064844,0.00022522255,0.00013279502,0.00009843796],"domain_scores_gemma":[0.99955493,0.00016280204,0.00003804755,0.000091246526,0.000108208405,0.000044764452],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012722622,0.0020480342,0.0015464997,0.0019511433,0.0005282754,0.0011481954,0.0034105335,0.0017961338,0.0072468766],"category_scores_gemma":[0.0024718235,0.000464227,0.0015876369,0.0013192301,0.00062942045,0.0026829473,0.002054155,0.0015073244,0.0030447736],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010725418,0.0006627087,0.0016160562,0.00050548813,0.0003622572,0.0003517152,0.00020994374,0.051569156,0.05345302,0.0043789884,0.040027615,0.8457906],"study_design_scores_gemma":[0.00010923715,0.000370326,0.0010090572,0.00003937401,0.00013464273,0.0003361755,0.00006895575,0.96400183,0.020622639,0.0065759523,0.006683908,0.000047867034],"about_ca_topic_score_codex":0.015196483,"about_ca_topic_score_gemma":0.01573423,"teacher_disagreement_score":0.015196483,"about_ca_system_score_codex":0.001564969,"about_ca_system_score_gemma":0.0011067764,"threshold_uncertainty_score":0.030216038},"labels":[],"label_agreement":null},{"id":"W4312458841","doi":"10.1109/cvpr52688.2022.00359","title":"Modular Action Concept Grounding in Semantic Video Prediction","year":2022,"lang":"en","type":"article","venue":"2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Vector Institute; University of Toronto","funders":"Canadian Institute for Advanced Research","keywords":"Computer science; Artificial intelligence; Action (physics); Task (project management); Generalization; Object (grammar); Modular design; Machine learning; Minimum bounding box; Exploit; Bounding overwatch; Natural language processing; Image (mathematics)","score_opus":0.05111951061090442,"score_gpt":0.3018798975331176,"score_spread":0.2507603869222132,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312458841","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11780752,0.0009708583,0.87338567,0.00065287226,0.000109540604,0.00011847235,0.0011235034,0.0036438867,0.002187692],"genre_scores_gemma":[0.84382486,0.00035200932,0.15079416,0.00024248521,0.00009285383,0.00012614676,0.0023948436,0.00012822433,0.0020444673],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994192,0.00012143046,0.00002002043,0.00028302305,0.00007996777,0.000076364464],"domain_scores_gemma":[0.9986406,0.00066775276,0.00016596733,0.00021543885,0.0002010512,0.00010913371],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011038544,0.0014093723,0.0008356865,0.0013971932,0.00039796752,0.00067846006,0.0020246713,0.0013106233,0.0017371807],"category_scores_gemma":[0.0034265055,0.00041562848,0.0007621889,0.0012652713,0.0009032871,0.002284661,0.0009846926,0.0018232594,0.0005635502],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006050847,0.00027158143,0.00815731,0.0001969136,0.00013776052,0.00020057557,0.00028019247,0.50324607,0.010120107,0.012052457,0.0085684825,0.4561635],"study_design_scores_gemma":[0.000008765457,0.000029363084,0.00049183273,0.000010926352,0.000010195549,0.000019761006,0.000013206181,0.98991525,0.0018206699,0.0072105825,0.00046248615,0.0000069568064],"about_ca_topic_score_codex":0.014912653,"about_ca_topic_score_gemma":0.015145336,"teacher_disagreement_score":0.014912653,"about_ca_system_score_codex":0.0015210239,"about_ca_system_score_gemma":0.00092764874,"threshold_uncertainty_score":0.029651701},"labels":[],"label_agreement":null},{"id":"W4312523916","doi":"10.1109/cvpr52688.2022.01509","title":"Multi-Modal Dynamic Graph Transformer for Visual Grounding","year":2022,"lang":"en","type":"article","venue":"2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Initialization; Transformer; Ground; Modal; Graph; Ground truth; Artificial intelligence; Data mining; Pattern recognition (psychology); Theoretical computer science; Voltage","score_opus":0.04175260766584023,"score_gpt":0.32924491239606796,"score_spread":0.28749230473022774,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312523916","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010964858,0.00038155916,0.97318345,0.0001420591,0.000058880258,0.000111234534,0.001000543,0.011582932,0.0025745262],"genre_scores_gemma":[0.3695062,0.0004903699,0.6138479,0.0004954483,0.000073661315,0.00023673661,0.007425601,0.0026646524,0.0052594496],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99924195,0.000110636676,0.000024812276,0.0003189913,0.00021858107,0.00008504671],"domain_scores_gemma":[0.9994306,0.00014731572,0.000055400826,0.0002120616,0.00010824985,0.000046315665],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00055249705,0.0015898485,0.0009540766,0.0025099455,0.0005661956,0.0014542048,0.0023293332,0.0013248489,0.007407507],"category_scores_gemma":[0.0024820666,0.0005845357,0.0015984838,0.0019623113,0.0010991902,0.003260154,0.0025935746,0.0016326521,0.0034739014],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039553217,0.0002333257,0.0017700861,0.0004533886,0.0001422515,0.0003226202,0.0003520366,0.16641326,0.05065317,0.028627655,0.028747525,0.72188914],"study_design_scores_gemma":[0.000044653665,0.00009826576,0.0007693599,0.00004070698,0.000037769383,0.00030473163,0.00021729205,0.9179779,0.020545423,0.046944674,0.012974923,0.00004427415],"about_ca_topic_score_codex":0.008455687,"about_ca_topic_score_gemma":0.017638978,"teacher_disagreement_score":0.008455687,"about_ca_system_score_codex":0.0011617701,"about_ca_system_score_gemma":0.00083196926,"threshold_uncertainty_score":0.024780631},"labels":[],"label_agreement":null},{"id":"W4312661653","doi":"10.1007/978-3-031-13064-9_24","title":"Caption and Observation Based on the Algorithm for Triangulation (COBALT): Preliminary Results from a Beta Trial","year":2022,"lang":"en","type":"book-chapter","venue":"Lecture notes in information systems and organisation","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"HEC Montréal","funders":"","keywords":"Suite; Workflow; Triangulation; Cobalt; Computer science; BETA (programming language); Software; Ecosystem; Simulation; Data science; Database; Geography; Ecology; Operating system; Archaeology; Cartography; Chemistry; Biology; Programming language","score_opus":0.02611223881160609,"score_gpt":0.24220170956167514,"score_spread":0.21608947075006904,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312661653","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02466229,0.0008338575,0.9388493,0.00037870923,0.0014239985,0.0010796038,0.0017180255,0.008339354,0.022714779],"genre_scores_gemma":[0.13173743,0.0006901948,0.8351841,0.00014526772,0.00022977778,0.0004626891,0.00406477,0.005377309,0.022108534],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977337,0.000682753,0.0001255869,0.00047723425,0.00081017095,0.00017048864],"domain_scores_gemma":[0.9868707,0.005606315,0.0002584581,0.0038451306,0.0030772574,0.00034214908],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002144311,0.0017495954,0.0020957133,0.001828472,0.0015866274,0.002942733,0.003105744,0.0025260383,0.0766596],"category_scores_gemma":[0.015757028,0.00092275714,0.0012627412,0.0033688564,0.0016915576,0.0031902138,0.0029637574,0.0022821429,0.01741207],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028603065,0.00033170218,0.0014522373,0.0007491599,0.00010238395,0.0006397698,0.0011580327,0.02899576,0.044067066,0.010604676,0.03604907,0.87298983],"study_design_scores_gemma":[0.0007544115,0.0017620047,0.006229513,0.0003596509,0.0004195322,0.0030728444,0.001842622,0.7086552,0.12588379,0.03328106,0.11733092,0.0004084228],"about_ca_topic_score_codex":0.006560771,"about_ca_topic_score_gemma":0.004514526,"teacher_disagreement_score":0.0766596,"about_ca_system_score_codex":0.0007812365,"about_ca_system_score_gemma":0.001503107,"threshold_uncertainty_score":0.2564519},"labels":[],"label_agreement":null},{"id":"W4313343280","doi":"10.1007/978-3-031-23028-8_29","title":"Refining AttnGAN Using Attention on Attention Network","year":2022,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Architecture; Relation (database); Function (biology); Artificial intelligence; Network architecture; Attention network; State (computer science); Mode (computer interface); Image (mathematics); Natural language processing; Data mining; Human–computer interaction; Programming language","score_opus":0.027820712809457675,"score_gpt":0.28309235193141985,"score_spread":0.25527163912196216,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4313343280","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.059169907,0.00093462353,0.9143053,0.00030447147,0.00029843187,0.00013374297,0.00027974322,0.007372013,0.017201606],"genre_scores_gemma":[0.676566,0.00044423604,0.289207,0.00034514783,0.0001315162,0.00008453598,0.0011191485,0.0012969424,0.030805428],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99937314,0.00010091995,0.000025629623,0.00019598893,0.00016794338,0.00013643599],"domain_scores_gemma":[0.9991393,0.00026540947,0.000039425526,0.0002128273,0.00029370532,0.0000492548],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007721128,0.0010216399,0.0010735671,0.001139782,0.0008865952,0.0009773048,0.0012140591,0.00094975403,0.01053005],"category_scores_gemma":[0.0023499192,0.0004013188,0.0007368104,0.00084361446,0.00053603615,0.0023682634,0.0013052573,0.0014462658,0.0020785886],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00053597504,0.00020239023,0.0023781809,0.00013083071,0.00006595736,0.00021993738,0.00013059497,0.087260015,0.028547976,0.019383915,0.0115060285,0.8496382],"study_design_scores_gemma":[0.000018317989,0.000066620014,0.000761505,0.000015174565,0.000049724556,0.00007668334,0.00004928709,0.9693925,0.0104637435,0.014691286,0.004400399,0.000014711957],"about_ca_topic_score_codex":0.015341019,"about_ca_topic_score_gemma":0.018355593,"teacher_disagreement_score":0.015341019,"about_ca_system_score_codex":0.0010107403,"about_ca_system_score_gemma":0.001100396,"threshold_uncertainty_score":0.035226524},"labels":[],"label_agreement":null},{"id":"W4313563617","doi":"10.1145/3551349.3560433","title":"Consistent Scene Graph Generation by Constraint Optimization","year":2022,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Scene graph; Local consistency; Consistency (knowledge bases); Constraint (computer-aided design); Graph; Probabilistic logic; Context (archaeology); Artificial intelligence; Theoretical computer science; Constraint satisfaction; Mathematics","score_opus":0.016020957269543702,"score_gpt":0.24106832676481343,"score_spread":0.22504736949526974,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4313563617","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01621727,0.00025225704,0.9713327,0.00032351856,0.000050756018,0.00024731047,0.001809112,0.0071648695,0.0026021749],"genre_scores_gemma":[0.16324514,0.00019635385,0.8190964,0.0003473056,0.000040460134,0.0003328431,0.01192682,0.0024282206,0.0023863933],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99737465,0.00070079934,0.0001213341,0.0008069235,0.0008585783,0.00013769807],"domain_scores_gemma":[0.9949609,0.0023889127,0.0003174475,0.0012953266,0.00091059424,0.00012689094],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017027986,0.0020812955,0.0011804457,0.002453327,0.0008634359,0.0019219704,0.0030095507,0.0016947243,0.005638662],"category_scores_gemma":[0.009344534,0.00086515915,0.0018433777,0.0023119838,0.0010979652,0.0029391004,0.002380769,0.0021523752,0.0019898915],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025683496,0.0002560378,0.0025131763,0.000504185,0.00017932305,0.00025829932,0.00017310149,0.6277435,0.01299425,0.020472355,0.027758416,0.30689055],"study_design_scores_gemma":[0.00003872665,0.000031070944,0.00031795353,0.000013227573,0.000020960071,0.00006980856,0.000048934162,0.97076404,0.005860617,0.018772922,0.0040434455,0.000018226898],"about_ca_topic_score_codex":0.010840816,"about_ca_topic_score_gemma":0.022154195,"teacher_disagreement_score":0.010840816,"about_ca_system_score_codex":0.001604096,"about_ca_system_score_gemma":0.0022614906,"threshold_uncertainty_score":0.021555424},"labels":[],"label_agreement":null},{"id":"W4313639341","doi":"10.1002/9781119898566.ch2","title":"Metaverse: The New North Star","year":2023,"lang":"en","type":"other","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institut National de la Recherche Scientifique","funders":"","keywords":"Metaverse; Computer science; Key (lock); Human–computer interaction; Virtual reality; Computer security","score_opus":0.015090409323806033,"score_gpt":0.2642368463415059,"score_spread":0.24914643701769987,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4313639341","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0092672175,0.009640779,0.05969407,0.035696488,0.010544782,0.000137884,0.0006301046,0.003628949,0.8707598],"genre_scores_gemma":[0.14548814,0.018121857,0.05475801,0.0110810585,0.0035919922,0.00038654363,0.0025387302,0.006572655,0.757461],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9980059,0.00045223383,0.00008199051,0.00028564414,0.0007972507,0.00037700514],"domain_scores_gemma":[0.9972963,0.00041227226,0.00009336406,0.0006091447,0.0007297324,0.0008593094],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003969019,0.00068691734,0.0004904863,0.0015794443,0.0055987267,0.011606335,0.0019402214,0.0040206388,0.05583754],"category_scores_gemma":[0.00523636,0.0005677146,0.0006770597,0.0015795609,0.004368803,0.026737187,0.0115071125,0.0054464173,0.024035],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000094653835,0.00004444853,0.0004347902,0.00016587744,0.000009225284,0.0002698895,0.0016071644,0.0005931559,0.0007100968,0.72108024,0.1824673,0.09252316],"study_design_scores_gemma":[0.000004489601,0.00001884178,0.00007564427,0.00013617838,0.0000020776708,0.00014023797,0.0005015541,0.00042540248,0.00029377695,0.043530773,0.9548601,0.000010851989],"about_ca_topic_score_codex":0.004265283,"about_ca_topic_score_gemma":0.0058003454,"teacher_disagreement_score":0.05583754,"about_ca_system_score_codex":0.0031761655,"about_ca_system_score_gemma":0.004865228,"threshold_uncertainty_score":0.18679518},"labels":[],"label_agreement":null},{"id":"W4317659260","doi":"10.1145/3580519","title":"Validation of an Improved Vision-Based Web Page Parsing Pipeline","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on the Web","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; McGill University; Mount Allison University","funders":"","keywords":"Computer science; Parsing; Pipeline (software); Segmentation; Ground truth; Web page; Set (abstract data type); Artificial intelligence; Pruning; Zoom; Machine learning; Visualization; Interface (matter); Information retrieval; World Wide Web; Programming language","score_opus":0.021552036406250807,"score_gpt":0.294845347279104,"score_spread":0.2732933108728532,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4317659260","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40872392,0.00080525497,0.54351276,0.00060040684,0.0003599232,0.001172968,0.0040414766,0.035886772,0.0048965546],"genre_scores_gemma":[0.5913982,0.00021485437,0.3909176,0.00044037122,0.000061385625,0.0006425056,0.010666197,0.0022859364,0.00337294],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99006325,0.0036026523,0.0006645191,0.0026179887,0.0024819584,0.0005696251],"domain_scores_gemma":[0.9743807,0.012181151,0.0010148302,0.0051086936,0.0067122546,0.0006024091],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009055144,0.0019247467,0.0015078175,0.002497596,0.0010603459,0.0044013476,0.0034702222,0.0038573733,0.0032710463],"category_scores_gemma":[0.036703404,0.0009351394,0.0012879947,0.0014653048,0.0013493462,0.0037636538,0.002461702,0.0026783529,0.0030705798],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0035658216,0.0027739783,0.019563526,0.0019992169,0.00066122017,0.00064946455,0.0017701727,0.29892257,0.15565944,0.005670241,0.018655464,0.4901089],"study_design_scores_gemma":[0.000090581925,0.0006098554,0.008153237,0.00011129237,0.000085994994,0.00024700075,0.00024876097,0.9072962,0.07641822,0.0026550195,0.003999457,0.00008431674],"about_ca_topic_score_codex":0.011579288,"about_ca_topic_score_gemma":0.008132906,"teacher_disagreement_score":0.011579288,"about_ca_system_score_codex":0.0019266851,"about_ca_system_score_gemma":0.0023399966,"threshold_uncertainty_score":0.047888696},"labels":[],"label_agreement":null},{"id":"W4318824050","doi":"10.1109/icdm54844.2022.00125","title":"Set2Box: Similarity Preserving Representation Learning for Sets","year":2022,"lang":"en","type":"article","venue":"2022 IEEE International Conference on Data Mining (ICDM)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Similarity (geometry); Computer science; ENCODE; Set (abstract data type); Representation (politics); Source code; Code (set theory); Theoretical computer science; Data mining; Artificial intelligence; Algorithm; Programming language","score_opus":0.23452707947196416,"score_gpt":0.419734007872125,"score_spread":0.18520692840016084,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4318824050","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010129099,0.0004462628,0.9835238,0.00014684565,0.00008366398,0.00013010435,0.00069534633,0.0036575126,0.0011873831],"genre_scores_gemma":[0.17395814,0.0005952276,0.81359416,0.00046436506,0.000115140494,0.00063239224,0.005885405,0.00049927586,0.0042559328],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99789596,0.00046417065,0.00014092783,0.0005168547,0.00085358333,0.00012847876],"domain_scores_gemma":[0.9980075,0.00065124134,0.00016453322,0.0007198547,0.0003757625,0.00008116497],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018413478,0.0014719645,0.0017078156,0.0019770344,0.00061585684,0.001935975,0.0034010992,0.0015432193,0.006299267],"category_scores_gemma":[0.007902052,0.0005405559,0.0013024117,0.0023865292,0.0008301297,0.0049938117,0.0037324543,0.0022618275,0.002755461],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047641163,0.00026941128,0.0011528723,0.00036051747,0.00013504979,0.00009540867,0.00019592632,0.13145582,0.011370154,0.03318512,0.019957535,0.8013457],"study_design_scores_gemma":[0.0000421691,0.00019024813,0.000345619,0.000041050997,0.00002214092,0.00011769528,0.00006444115,0.95380574,0.009758161,0.028586382,0.0069880104,0.000038354196],"about_ca_topic_score_codex":0.0029787675,"about_ca_topic_score_gemma":0.003546347,"teacher_disagreement_score":0.006299267,"about_ca_system_score_codex":0.0012613846,"about_ca_system_score_gemma":0.0013730328,"threshold_uncertainty_score":0.021073103},"labels":[],"label_agreement":null},{"id":"W4319323278","doi":"10.48550/arxiv.2302.01403","title":"Self-Supervised Relation Alignment for Scene Graph Generation","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Compute Canada; Natural Sciences and Engineering Research Council of Canada; Government of Canada; Canadian Institute for Advanced Research","keywords":"Computer science; Scene graph; Graph; Message passing; Artificial intelligence; Regularization (linguistics); Machine learning; Pattern recognition (psychology); Theoretical computer science; Parallel computing","score_opus":0.11173321987582735,"score_gpt":0.22186852859608186,"score_spread":0.11013530872025451,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319323278","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021491727,0.00010097152,0.9691928,0.00013704444,0.00003553274,0.000059799575,0.00019520413,0.0072954255,0.0014914294],"genre_scores_gemma":[0.41495076,0.000109114226,0.5761488,0.0002904423,0.00005636326,0.00013803504,0.001937444,0.0012774003,0.0050916397],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99949265,0.00012428236,0.000013132299,0.00022158578,0.00010805791,0.000040311344],"domain_scores_gemma":[0.99919266,0.00026038795,0.00008762206,0.0002836051,0.00013041846,0.000045200642],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005993202,0.000865336,0.0005550655,0.0006739467,0.0004126187,0.00050941744,0.0020404768,0.0009176614,0.0042735008],"category_scores_gemma":[0.0021276707,0.00042318733,0.00069595804,0.0005921605,0.0006159252,0.001708993,0.0011438341,0.0015529061,0.0017246824],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032162358,0.00029064107,0.00208594,0.00020269594,0.00009470352,0.00019182377,0.00020435722,0.3416875,0.06046773,0.026571607,0.014723472,0.5531579],"study_design_scores_gemma":[0.000008295838,0.000029206753,0.00019435151,0.000003582828,0.000006288874,0.000037715836,0.000013665764,0.98178124,0.009095938,0.0076741865,0.0011499329,0.0000055175933],"about_ca_topic_score_codex":0.0026124823,"about_ca_topic_score_gemma":0.0074564503,"teacher_disagreement_score":0.0042735008,"about_ca_system_score_codex":0.0006798756,"about_ca_system_score_gemma":0.0007043029,"threshold_uncertainty_score":0.014296293},"labels":[],"label_agreement":null},{"id":"W4319971251","doi":"10.3390/a16020097","title":"Nemesis: Neural Mean Teacher Learning-Based Emotion-Centric Speaker","year":2023,"lang":"en","type":"article","venue":"Algorithms","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Laurentian University","funders":"","keywords":"Closed captioning; Computer science; Natural language processing; Artificial intelligence; Task (project management); Natural language; Principle of maximum entropy; Image (mathematics); Perceptron; Artificial neural network; Speech recognition","score_opus":0.017980481412018783,"score_gpt":0.2741770904438704,"score_spread":0.2561966090318516,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319971251","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.060855825,0.0026834744,0.9098915,0.00074792886,0.00088666886,0.00030547086,0.0013430621,0.015532381,0.0077537387],"genre_scores_gemma":[0.6105275,0.0011472849,0.34396598,0.0012513483,0.00065345864,0.0006679209,0.007028749,0.0011510581,0.03360661],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99951255,0.00014611968,0.000017743816,0.00019184954,0.00007501052,0.000056600824],"domain_scores_gemma":[0.9994667,0.00025426745,0.000033185424,0.00009142139,0.00011760185,0.000036815894],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011349272,0.0013680402,0.00092164084,0.0005636762,0.0004621177,0.0006755267,0.0019426107,0.001220847,0.004277906],"category_scores_gemma":[0.0027267132,0.00036092795,0.0011106842,0.00043704166,0.0006254099,0.0013576063,0.0014338121,0.0022296198,0.0024241428],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011158059,0.00048765336,0.0016488325,0.00034854456,0.00028555762,0.00033943972,0.00035265996,0.17747231,0.034011394,0.005550208,0.03301105,0.7453765],"study_design_scores_gemma":[0.000047584046,0.00016830444,0.0004244725,0.000017594315,0.00004038228,0.00009125342,0.00004532113,0.9836053,0.008315894,0.0032414566,0.00398089,0.000021575022],"about_ca_topic_score_codex":0.00369346,"about_ca_topic_score_gemma":0.005884267,"teacher_disagreement_score":0.004277906,"about_ca_system_score_codex":0.0008161375,"about_ca_system_score_gemma":0.0007226748,"threshold_uncertainty_score":0.014311075},"labels":[],"label_agreement":null},{"id":"W4320481739","doi":"10.1007/978-3-031-25069-9_15","title":"Distinctive Image Captioning via CLIP Guided Group Optimization","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Closed captioning; Optimal distinctiveness theory; Computer science; Image (mathematics); Artificial intelligence; Metric (unit); Natural language processing; Embedding; Ground truth; Focus (optics); Baseline (sea); Consistency (knowledge bases); Pattern recognition (psychology); Machine learning; Information retrieval","score_opus":0.017930149582978804,"score_gpt":0.27462452038326973,"score_spread":0.25669437080029095,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4320481739","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0064592496,0.00018041182,0.984083,0.0001457842,0.00021289317,0.00008624028,0.00016053293,0.0029622803,0.005709596],"genre_scores_gemma":[0.13224731,0.0003141339,0.8469424,0.0003525675,0.0003355688,0.00021939054,0.0013046492,0.0019503206,0.01633372],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99947697,0.00010217345,0.000015593183,0.00015084438,0.00017787177,0.00007651599],"domain_scores_gemma":[0.9994104,0.00018239248,0.000036428468,0.00016797976,0.00014499042,0.000057762263],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005304999,0.0014985022,0.0013210941,0.0011510004,0.0006060549,0.0013155644,0.0016309773,0.0015150929,0.018344942],"category_scores_gemma":[0.0015848962,0.0005464517,0.0012359134,0.0014590493,0.0007430146,0.0013076128,0.0019204781,0.0019905004,0.0064101624],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005357747,0.00017736279,0.00017951598,0.00027238182,0.00008788447,0.00025792173,0.00015664667,0.10341656,0.08551177,0.017201219,0.044967744,0.7472351],"study_design_scores_gemma":[0.000031791238,0.000102398226,0.00014153465,0.0000124625285,0.000027531212,0.00015434137,0.00004963399,0.95748526,0.021929054,0.011215586,0.00882696,0.000023406743],"about_ca_topic_score_codex":0.0012414687,"about_ca_topic_score_gemma":0.0017877138,"teacher_disagreement_score":0.018344942,"about_ca_system_score_codex":0.0004241556,"about_ca_system_score_gemma":0.0005245101,"threshold_uncertainty_score":0.061369956},"labels":[],"label_agreement":null},{"id":"W4321608662","doi":"10.36227/techrxiv.22133894.v1","title":"Image Captioning for the Visually Impaired and Blind: A Recipe for Low-Resource Languages","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Closed captioning; Visually impaired; Computer science; Assistive technology; Hearing impaired; Human–computer interaction; Resource (disambiguation); Artificial intelligence; Computer vision; Speech recognition; Natural language processing; Image (mathematics); Audiology","score_opus":0.03894768839832212,"score_gpt":0.3599525522228245,"score_spread":0.3210048638245024,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4321608662","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041561767,0.003073598,0.89766896,0.0057336763,0.0014445835,0.0011031778,0.001175653,0.015948232,0.03229037],"genre_scores_gemma":[0.22808312,0.0033413197,0.74146193,0.0018279994,0.0005109404,0.0007358573,0.001640051,0.0016917496,0.020706985],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99960977,0.00015828758,0.000028742335,0.000056434274,0.00010831873,0.000038489274],"domain_scores_gemma":[0.9982126,0.0005554768,0.00009494138,0.0003613612,0.000633386,0.00014220117],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009738504,0.0013211658,0.000452376,0.00084405026,0.0008847749,0.0017093639,0.00088203343,0.0013894989,0.011790787],"category_scores_gemma":[0.0043996805,0.0002465159,0.0005087951,0.00042538572,0.0009834793,0.0026746362,0.0014759155,0.0010949997,0.00584759],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007487987,0.00021993669,0.0011029615,0.0021577645,0.00004999853,0.0014620327,0.003555383,0.004333822,0.18005589,0.013613474,0.08601889,0.70668113],"study_design_scores_gemma":[0.00017851118,0.001342385,0.008426102,0.0016242913,0.00027321733,0.010402606,0.006704146,0.08587207,0.31096178,0.057447217,0.516218,0.0005497058],"about_ca_topic_score_codex":0.0008329105,"about_ca_topic_score_gemma":0.001270349,"teacher_disagreement_score":0.011790787,"about_ca_system_score_codex":0.00036775728,"about_ca_system_score_gemma":0.00046938463,"threshold_uncertainty_score":0.03944409},"labels":[],"label_agreement":null},{"id":"W4327523052","doi":"10.1109/access.2023.3257280","title":"Video Relationship Detection Using Mixture of Experts","year":2023,"lang":"en","type":"article","venue":"IEEE Access","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Microsoft Research Asia; Microsoft Research; Microsoft","keywords":"Computer science; Artificial intelligence; Inference; Machine learning; Classifier (UML); Artificial neural network; Predicate (mathematical logic); Object detection; Pattern recognition (psychology)","score_opus":0.07794275924308183,"score_gpt":0.37545058696163475,"score_spread":0.2975078277185529,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4327523052","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03278427,0.00040589203,0.96377677,0.00023536573,0.00003532387,0.00008685922,0.00012125945,0.0010400383,0.0015142378],"genre_scores_gemma":[0.5874358,0.0003749886,0.40612897,0.0004118662,0.000120166806,0.00012895202,0.0006797138,0.0001360879,0.00458343],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99848527,0.0003117986,0.00005840219,0.0005981405,0.00031761732,0.00022879845],"domain_scores_gemma":[0.99788576,0.0010202874,0.00027864473,0.0002135612,0.00040989576,0.00019180289],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019432766,0.0012256294,0.001257834,0.0022427868,0.0004975617,0.0013571351,0.0029140234,0.0026188267,0.0018063595],"category_scores_gemma":[0.006195415,0.00085659616,0.0014518456,0.0010498936,0.00066572987,0.0024532322,0.0020697992,0.0020642332,0.0007545653],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00093673216,0.00029982405,0.0058383653,0.00017598049,0.00029729932,0.0006306156,0.00036848753,0.36686322,0.019274045,0.012818226,0.0056556356,0.5868415],"study_design_scores_gemma":[0.000005639178,0.00002064613,0.00028061806,0.000006088039,0.000009580497,0.00005283932,0.000014878383,0.99428236,0.0020981242,0.0028488876,0.00037347912,0.0000068790346],"about_ca_topic_score_codex":0.0071477825,"about_ca_topic_score_gemma":0.0070026387,"teacher_disagreement_score":0.0071477825,"about_ca_system_score_codex":0.0012744969,"about_ca_system_score_gemma":0.0008326426,"threshold_uncertainty_score":0.01421237},"labels":[],"label_agreement":null},{"id":"W4328029477","doi":"10.1111/cgf.14668","title":"A Drone Video Clip Dataset and its Applications in Automated Cinematography","year":2022,"lang":"en","type":"article","venue":"Computer Graphics Forum","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Korea Creative Content Agency","keywords":"Drone; Computer science; Computer vision; Artificial intelligence; Cinematography; CLIPS; Video capture; Computer graphics (images); Video processing","score_opus":0.010880231189374133,"score_gpt":0.2707722317890653,"score_spread":0.25989200059969114,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4328029477","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3259149,0.007378282,0.025561124,0.00095510477,0.0013119116,0.0017894361,0.5951214,0.025179857,0.016787985],"genre_scores_gemma":[0.14732088,0.0011457532,0.034260288,0.00015141715,0.00018123712,0.00034038638,0.81281537,0.00037954454,0.003405205],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99915266,0.00009982019,0.000075511314,0.00029542847,0.00027049612,0.000106035746],"domain_scores_gemma":[0.9988637,0.00025548626,0.00008301269,0.00031177315,0.00035103975,0.00013509467],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005271424,0.0018353548,0.00070926436,0.00490792,0.000777742,0.0009012755,0.0013748463,0.001321875,0.0047421763],"category_scores_gemma":[0.0023107312,0.0002313597,0.0008689463,0.0025902204,0.00038056163,0.0011144122,0.00086137,0.0009828649,0.0029201466],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001462219,0.0016613384,0.017131165,0.004207512,0.0005035638,0.0027872534,0.000648762,0.023951238,0.031597827,0.0021635487,0.54857206,0.3653135],"study_design_scores_gemma":[0.0005587211,0.001145364,0.16637616,0.0010157411,0.00035092636,0.004502296,0.0031457362,0.30441517,0.054551456,0.0028535023,0.46069518,0.00038976182],"about_ca_topic_score_codex":0.03238658,"about_ca_topic_score_gemma":0.061427053,"teacher_disagreement_score":0.03238658,"about_ca_system_score_codex":0.0008804913,"about_ca_system_score_gemma":0.0005531747,"threshold_uncertainty_score":0.06439614},"labels":[],"label_agreement":null},{"id":"W4361193744","doi":"10.48550/arxiv.2303.14979","title":"Lexicon-Enhanced Self-Supervised Training for Multilingual Dense Retrieval","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Relevance (law); Artificial intelligence; Lexicon; Generator (circuit theory); Training set; Labeled data; Natural language processing; Information retrieval; Machine learning","score_opus":0.14579180794735194,"score_gpt":0.2567737515440782,"score_spread":0.11098194359672628,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4361193744","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.081748486,0.001316412,0.89404005,0.00034408204,0.00011613545,0.00032907916,0.0010543861,0.016566891,0.0044844933],"genre_scores_gemma":[0.6100237,0.00043996697,0.36802003,0.0008751039,0.00020784153,0.0006387017,0.010443203,0.0009799816,0.008371418],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987041,0.0004974217,0.00009105529,0.00038800103,0.00020371226,0.00011573385],"domain_scores_gemma":[0.9970174,0.0014318086,0.00015843892,0.0006705384,0.00062221655,0.00009959804],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018793062,0.0012545295,0.0014056219,0.0015739676,0.00067379134,0.00071440917,0.002415828,0.0012218696,0.0033039511],"category_scores_gemma":[0.006302414,0.0006239262,0.0009440113,0.0015626566,0.0009083205,0.002746833,0.0017708441,0.0018443378,0.003098919],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059035176,0.00092897116,0.004337232,0.00044555907,0.0001920366,0.00027974983,0.00037814665,0.11162988,0.027049625,0.00403878,0.02838827,0.82174146],"study_design_scores_gemma":[0.00006711134,0.00014847357,0.0006631144,0.0000172216,0.00003283456,0.00016142544,0.000084955944,0.98141897,0.009089881,0.0050450796,0.0032423644,0.000028676199],"about_ca_topic_score_codex":0.0065143188,"about_ca_topic_score_gemma":0.0138970325,"teacher_disagreement_score":0.0065143188,"about_ca_system_score_codex":0.00077606144,"about_ca_system_score_gemma":0.0014861037,"threshold_uncertainty_score":0.012952805},"labels":[],"label_agreement":null},{"id":"W4364322545","doi":"10.1109/tai.2023.3266183","title":"Visual Relationship Detection for Workplace Safety Applications","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Bayesian network; Context (archaeology); Object detection; Visualization; Artificial neural network; Artificial intelligence; Object (grammar); Graphical user interface; Machine learning; Data mining; Pattern recognition (psychology); Programming language","score_opus":0.05401277960001292,"score_gpt":0.34603996480937,"score_spread":0.29202718520935705,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4364322545","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0724835,0.0013878185,0.86038715,0.0017477156,0.0003561691,0.00051325397,0.003219207,0.04571884,0.014186263],"genre_scores_gemma":[0.55805737,0.00055405434,0.4257096,0.00086823176,0.0001129293,0.0003220178,0.0049750297,0.0007196627,0.008681071],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988009,0.00020397891,0.000049423383,0.00037557288,0.00047213965,0.00009803091],"domain_scores_gemma":[0.9981281,0.00058188284,0.00014667152,0.0002845835,0.00074786594,0.00011091669],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013575482,0.0010540261,0.00047132664,0.0014058072,0.0004419376,0.0012672795,0.002095411,0.0018508743,0.012549068],"category_scores_gemma":[0.0079668965,0.0003876494,0.00062846794,0.0006326187,0.00030353022,0.0018305794,0.0018137402,0.0010988713,0.0048738746],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00094212336,0.00059652166,0.008517876,0.00048258778,0.00009922912,0.00031935197,0.00020481765,0.03259439,0.039283436,0.002977534,0.036288988,0.8776932],"study_design_scores_gemma":[0.0000762973,0.00043867517,0.008535148,0.00015092191,0.00006830465,0.0004598184,0.00021998122,0.8804022,0.06550353,0.020654175,0.023418047,0.00007296047],"about_ca_topic_score_codex":0.0054332986,"about_ca_topic_score_gemma":0.008523021,"teacher_disagreement_score":0.012549068,"about_ca_system_score_codex":0.0009069036,"about_ca_system_score_gemma":0.00080432306,"threshold_uncertainty_score":0.041980803},"labels":[],"label_agreement":null},{"id":"W4367672600","doi":"10.36227/techrxiv.22133894.v2","title":"Image Captioning for the Visually Impaired and Blind: A Recipe for Low-Resource Languages","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Closed captioning; Computer science; Visually impaired; Assistive technology; Hearing impaired; Human–computer interaction; Artificial intelligence; Speech recognition; Computer vision; Image (mathematics); Natural language processing; Audiology","score_opus":0.03894768839832212,"score_gpt":0.3599525522228245,"score_spread":0.3210048638245024,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4367672600","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041561767,0.003073598,0.89766896,0.0057336763,0.0014445835,0.0011031778,0.001175653,0.015948232,0.03229037],"genre_scores_gemma":[0.22808312,0.0033413197,0.74146193,0.0018279994,0.0005109404,0.0007358573,0.001640051,0.0016917496,0.020706985],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99960977,0.00015828758,0.000028742335,0.000056434274,0.00010831873,0.000038489274],"domain_scores_gemma":[0.9982126,0.0005554768,0.00009494138,0.0003613612,0.000633386,0.00014220117],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009738504,0.0013211658,0.000452376,0.00084405026,0.0008847749,0.0017093639,0.00088203343,0.0013894989,0.011790787],"category_scores_gemma":[0.0043996805,0.0002465159,0.0005087951,0.00042538572,0.0009834793,0.0026746362,0.0014759155,0.0010949997,0.00584759],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007487987,0.00021993669,0.0011029615,0.0021577645,0.00004999853,0.0014620327,0.003555383,0.004333822,0.18005589,0.013613474,0.08601889,0.70668113],"study_design_scores_gemma":[0.00017851118,0.001342385,0.008426102,0.0016242913,0.00027321733,0.010402606,0.006704146,0.08587207,0.31096178,0.057447217,0.516218,0.0005497058],"about_ca_topic_score_codex":0.0008329105,"about_ca_topic_score_gemma":0.001270349,"teacher_disagreement_score":0.011790787,"about_ca_system_score_codex":0.00036775728,"about_ca_system_score_gemma":0.00046938463,"threshold_uncertainty_score":0.03944409},"labels":[],"label_agreement":null},{"id":"W4375854264","doi":"10.1109/cbs55922.2023.10115355","title":"Cross-modal Task Understanding and Execution of Voice-fingertip Reading Instruction by Using Small Family Service Robotic","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Task (project management); Robot; Service robot; Service (business); Artificial intelligence; Modal; Object (grammar); Human–computer interaction; Speech recognition; Natural language processing; Computer vision","score_opus":0.06714828239946975,"score_gpt":0.3062395290434044,"score_spread":0.23909124664393466,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4375854264","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20400839,0.0002395356,0.781854,0.0001519184,0.000034403616,0.00026753277,0.00013040053,0.0068500927,0.0064638057],"genre_scores_gemma":[0.8061704,0.00011222707,0.18830761,0.00008478391,0.000012673123,0.00015473914,0.00025437027,0.00013473183,0.004768443],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99956197,0.00006216514,0.000022253776,0.000167859,0.000110968664,0.000074789925],"domain_scores_gemma":[0.9997304,0.00006510402,0.000039834307,0.000056788092,0.00007093003,0.000036991183],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00036244857,0.00073040905,0.00042765835,0.00032435977,0.00038433296,0.00045807418,0.0007980011,0.00057509576,0.00238816],"category_scores_gemma":[0.0010415313,0.00016188597,0.00044423042,0.00015228301,0.00041462667,0.0011059664,0.00085655873,0.0005598534,0.0006568487],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00083012454,0.00066246575,0.007855966,0.00028606076,0.00007210215,0.00097641605,0.0021022663,0.043641955,0.24308321,0.005677925,0.0037079102,0.69110364],"study_design_scores_gemma":[0.000038342037,0.0007770934,0.013183629,0.00003179879,0.00008000264,0.00078069494,0.0009363806,0.8378691,0.12917672,0.0067216023,0.0102887,0.00011590689],"about_ca_topic_score_codex":0.0068546142,"about_ca_topic_score_gemma":0.007170297,"teacher_disagreement_score":0.0068546142,"about_ca_system_score_codex":0.00040407295,"about_ca_system_score_gemma":0.00093218597,"threshold_uncertainty_score":0.0136294365},"labels":[],"label_agreement":null},{"id":"W4376958607","doi":"10.32473/flairs.36.133328","title":"Towards a multi-modal Deep Learning Architecture for User Modeling","year":2023,"lang":"en","type":"article","venue":"Proceedings of the ... International Florida Artificial Intelligence Research Society Conference","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal; École de Technologie Supérieure","funders":"","keywords":"Computer science; Deep learning; Artificial intelligence; Modal; Convolutional neural network; User modeling; Representation (politics); Feature (linguistics); Feature learning; Machine learning; Architecture; Human–computer interaction; User interface","score_opus":0.16179346554749333,"score_gpt":0.40432547208867836,"score_spread":0.24253200654118504,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4376958607","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022610232,0.0003390938,0.97380674,0.000477485,0.000042797605,0.000039799263,0.00023063028,0.0011635986,0.0012895835],"genre_scores_gemma":[0.76288676,0.00051513733,0.22744545,0.00064740627,0.00007632315,0.00019338811,0.0007326132,0.0001272752,0.0073756743],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99962294,0.00012964213,0.000017117583,0.00011641889,0.000058986687,0.000054798056],"domain_scores_gemma":[0.99963367,0.00013430005,0.00003397321,0.000051252297,0.00011195769,0.000034741934],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00079879427,0.0007565875,0.00050384225,0.00050737895,0.00025492522,0.00068823475,0.0010892267,0.00088547793,0.0018059321],"category_scores_gemma":[0.0014493929,0.00047123342,0.00094536314,0.000433635,0.00042884512,0.0013295062,0.0013061335,0.0019902305,0.00074101135],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005907166,0.0005140891,0.007946049,0.00017592695,0.0003471581,0.0002038042,0.00058718,0.46765044,0.035286587,0.0210798,0.00882424,0.45679402],"study_design_scores_gemma":[0.0000023657794,0.000020847398,0.00029452035,0.000005527605,0.000009233243,0.000015730575,0.000010464521,0.99455786,0.0009756915,0.003666206,0.00043563874,0.000005945],"about_ca_topic_score_codex":0.0077880863,"about_ca_topic_score_gemma":0.01303585,"teacher_disagreement_score":0.0077880863,"about_ca_system_score_codex":0.00089272996,"about_ca_system_score_gemma":0.0006601407,"threshold_uncertainty_score":0.015485525},"labels":[],"label_agreement":null},{"id":"W4378419434","doi":"10.1007/978-3-031-33380-4_19","title":"TCR: Short Video Title Generation and Cover Selection with Attention Refinement","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Closed captioning; Selection (genetic algorithm); Cover (algebra); Task (project management); Popularity; Information retrieval; Focus (optics); Quality (philosophy); Multimedia; Artificial intelligence; Image (mathematics)","score_opus":0.024619940049163257,"score_gpt":0.2689565636779841,"score_spread":0.24433662362882083,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378419434","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016266486,0.0016775702,0.6562995,0.000850186,0.0032561137,0.0015172126,0.02974268,0.20791353,0.082476765],"genre_scores_gemma":[0.09777057,0.0011122351,0.6600516,0.00063121814,0.0010747556,0.001234955,0.07325686,0.016371258,0.14849661],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9995177,0.000047040285,0.000026626753,0.00015359037,0.00016899567,0.00008620119],"domain_scores_gemma":[0.99912363,0.0001619206,0.000035813322,0.00014516088,0.00041005487,0.00012337233],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00035172165,0.0020257593,0.0009959969,0.002780272,0.0006786447,0.001386295,0.0015460205,0.0011083948,0.1689541],"category_scores_gemma":[0.002165177,0.00059427967,0.0010275798,0.0018407519,0.00022695807,0.0015391917,0.0016037195,0.0008776205,0.0838853],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006018125,0.00008674198,0.00029175862,0.00031675567,0.00003216326,0.00023722296,0.00005050056,0.0016029235,0.03465245,0.0015235356,0.33759204,0.62301207],"study_design_scores_gemma":[0.00045293325,0.0005920947,0.0038886277,0.00025185934,0.00020554749,0.001048507,0.00047573674,0.24224347,0.17205523,0.01009703,0.5684967,0.00019222144],"about_ca_topic_score_codex":0.0072518536,"about_ca_topic_score_gemma":0.010774016,"teacher_disagreement_score":0.1689541,"about_ca_system_score_codex":0.00064824376,"about_ca_system_score_gemma":0.00086993707,"threshold_uncertainty_score":0.5652078},"labels":[],"label_agreement":null},{"id":"W4378465251","doi":"10.48550/arxiv.2305.14998","title":"An Examination of the Robustness of Reference-Free Image Captioning Evaluation Metrics","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Canadian Institute for Advanced Research; Nvidia","keywords":"Closed captioning; Robustness (evolution); Computer science; Artificial intelligence; Computer vision; Image (mathematics); Natural language processing","score_opus":0.1578930036878109,"score_gpt":0.25725857490790355,"score_spread":0.09936557122009265,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378465251","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.61167425,0.02032951,0.3023904,0.0016256514,0.0026604105,0.0022503138,0.010359614,0.022415543,0.026294328],"genre_scores_gemma":[0.82570964,0.0010246362,0.15318379,0.00048088652,0.00033505497,0.00074875995,0.013927877,0.0018759753,0.0027133822],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.96682924,0.0147081595,0.003606573,0.004947852,0.008962906,0.00094521727],"domain_scores_gemma":[0.855915,0.08720976,0.008437557,0.017159684,0.028970197,0.0023077866],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.033192944,0.0036736624,0.0014866713,0.008954925,0.0014205667,0.0044544,0.00272156,0.002668935,0.0025753127],"category_scores_gemma":[0.15128626,0.00052651874,0.0011672882,0.00400782,0.0016514166,0.004975601,0.003771692,0.0025748792,0.0012996899],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003939934,0.0008635824,0.04991807,0.0046441215,0.0025403458,0.00048059504,0.0025361702,0.087952845,0.044226658,0.0069367546,0.048214324,0.74774665],"study_design_scores_gemma":[0.00034104413,0.004936293,0.104594454,0.0010572871,0.0009126396,0.0018251662,0.0023964178,0.7087789,0.12090725,0.015968982,0.037501816,0.0007797746],"about_ca_topic_score_codex":0.0066816164,"about_ca_topic_score_gemma":0.0052423887,"teacher_disagreement_score":0.96680707,"about_ca_system_score_codex":0.002437448,"about_ca_system_score_gemma":0.0011654506,"threshold_uncertainty_score":0.17554319},"labels":[],"label_agreement":null},{"id":"W4379522641","doi":"10.21428/594757db.e0f8ffcd","title":"Guided Learning of Human Sensor Models with Low-Level Grounding","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada; Queen's University","keywords":"Interpretability; Representation (politics); Computer science; Artificial intelligence; Machine learning; Raw data; Pattern recognition (psychology)","score_opus":0.07235375822604777,"score_gpt":0.3253710443923826,"score_spread":0.25301728616633484,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4379522641","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018899543,0.000113580536,0.9776505,0.00016934478,0.000021689655,0.000036190628,0.00013324757,0.0023157783,0.00066023436],"genre_scores_gemma":[0.69184285,0.00016468545,0.30371678,0.00036923692,0.000041784675,0.00016460319,0.0013237412,0.0003539334,0.0020223854],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99908364,0.00024961642,0.000031617114,0.00038926819,0.00013740477,0.00010842363],"domain_scores_gemma":[0.99824715,0.00075428205,0.00018695313,0.00055029546,0.00017388116,0.000087493325],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010717838,0.0015503656,0.0015049895,0.000735459,0.0005251927,0.0015752345,0.0023495874,0.0017317522,0.0022877958],"category_scores_gemma":[0.005310956,0.0010692517,0.0015509807,0.0009454114,0.0017320779,0.0027704623,0.0030979086,0.003117491,0.0009736662],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001968525,0.00018549795,0.0024333925,0.00011550156,0.00011459703,0.00012720619,0.00028122234,0.75646436,0.010361723,0.008566661,0.0028845556,0.21826847],"study_design_scores_gemma":[0.0000053990443,0.000036393452,0.00022890182,0.000008186211,0.0000063814864,0.00002070838,0.000020644413,0.98802465,0.002047243,0.009271467,0.00032284795,0.0000072516546],"about_ca_topic_score_codex":0.006159606,"about_ca_topic_score_gemma":0.011668882,"teacher_disagreement_score":0.006159606,"about_ca_system_score_codex":0.0010892728,"about_ca_system_score_gemma":0.0012109338,"threshold_uncertainty_score":0.012247503},"labels":[],"label_agreement":null},{"id":"W4383560362","doi":"10.54254/2755-2721/5/20230511","title":"To describe the content of image: The view from image captioning","year":2023,"lang":"en","type":"article","venue":"Applied and Computational Engineering","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Closed captioning; Computer science; Natural language; Field (mathematics); Artificial intelligence; Natural language processing; Task (project management); Image (mathematics); Domain (mathematical analysis); Perspective (graphical); Engineering","score_opus":0.018034436956163823,"score_gpt":0.234893111473301,"score_spread":0.21685867451713717,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4383560362","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0075828917,0.059769116,0.8747844,0.017920023,0.0030962122,0.00039655747,0.0015879493,0.0030751657,0.031787723],"genre_scores_gemma":[0.101666376,0.0503032,0.8073982,0.0065960907,0.004210678,0.0005382861,0.0043745185,0.0018636347,0.023049038],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99823844,0.00077801873,0.00010337261,0.0003512063,0.00045216925,0.000076743396],"domain_scores_gemma":[0.9936207,0.0031087012,0.00031011534,0.001330308,0.0014058615,0.00022431166],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025801002,0.0015613433,0.0008020286,0.0028913154,0.0006989947,0.0049566845,0.0020233565,0.002552517,0.00727852],"category_scores_gemma":[0.01337814,0.0006143539,0.00085083523,0.0021686587,0.0038005637,0.009760413,0.0032419472,0.004668877,0.0045389077],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023655599,0.00006516411,0.0007651845,0.0028806303,0.00008398047,0.00024757895,0.0016612356,0.008724152,0.016451664,0.11803256,0.08857596,0.7622753],"study_design_scores_gemma":[0.00004418105,0.00028018406,0.0021402554,0.0015748375,0.00011473144,0.0028599086,0.0018095316,0.08043665,0.048978757,0.23532261,0.6262411,0.00019720841],"about_ca_topic_score_codex":0.0019479304,"about_ca_topic_score_gemma":0.0018993284,"teacher_disagreement_score":0.00727852,"about_ca_system_score_codex":0.0014832506,"about_ca_system_score_gemma":0.0008972445,"threshold_uncertainty_score":0.024349093},"labels":[],"label_agreement":null},{"id":"W4385565354","doi":"10.18653/v1/2023.acl-long.526","title":"Learning to Imagine: Visually-Augmented Natural Language Generation","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"National Natural Science Foundation of China","keywords":"Computer science; Paragraph; Natural language; Artificial intelligence; Transformer; Augmented reality; Sentence; Natural language processing; Code (set theory); Human–computer interaction; Programming language; World Wide Web","score_opus":0.01060861406612697,"score_gpt":0.3164279157877798,"score_spread":0.30581930172165284,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385565354","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024106473,0.00021474331,0.9480213,0.00021265844,0.0001600612,0.00014187346,0.00048103844,0.02053905,0.0061228373],"genre_scores_gemma":[0.4382744,0.00018741272,0.54812044,0.00030372603,0.000058390022,0.0001947958,0.0019767517,0.0012099098,0.009674129],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997054,0.00006990744,0.000011660455,0.00010911325,0.00007426798,0.00002962527],"domain_scores_gemma":[0.99950635,0.00022204724,0.00002954495,0.00013290645,0.00006935761,0.00003979926],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00051351794,0.000983484,0.0003204125,0.00034026438,0.00022061434,0.0008362318,0.001380619,0.0007482864,0.008078459],"category_scores_gemma":[0.0021911927,0.00031022166,0.00074913836,0.00018054542,0.0005435326,0.0017219097,0.0014001749,0.0011231963,0.0021659867],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00062113436,0.00035617824,0.0012798428,0.00046103398,0.00011812811,0.00061434536,0.0007792866,0.07324585,0.14453843,0.02184685,0.024998652,0.7311403],"study_design_scores_gemma":[0.000097274096,0.00026108325,0.0007886295,0.000032252407,0.00004870154,0.0005020989,0.00016774262,0.8616497,0.088134095,0.026085634,0.022176277,0.000056520097],"about_ca_topic_score_codex":0.0015846231,"about_ca_topic_score_gemma":0.0027932834,"teacher_disagreement_score":0.008078459,"about_ca_system_score_codex":0.00037984323,"about_ca_system_score_gemma":0.00038590922,"threshold_uncertainty_score":0.027025104},"labels":[],"label_agreement":null},{"id":"W4385566910","doi":"10.18653/v1/2023.semeval-1.281","title":"UAlberta at SemEval-2023 Task 1: Context Augmentation and Translation for Multilingual Visual Word Sense Disambiguation","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"SemEval; Computer science; Natural language processing; Task (project management); Encoder; Artificial intelligence; Context (archaeology); Word (group theory); Machine translation; Code (set theory); Word-sense disambiguation; Set (abstract data type); Text segmentation; Test set; Segmentation; Linguistics; Programming language; WordNet","score_opus":0.03527268440563371,"score_gpt":0.34970334595647884,"score_spread":0.31443066155084515,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385566910","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03917024,0.0068322085,0.15375572,0.0024858767,0.0039767106,0.002065219,0.21104008,0.49651846,0.0841554],"genre_scores_gemma":[0.09984785,0.0010626737,0.31227973,0.001569099,0.00032924858,0.0016777527,0.51927716,0.029129416,0.0348271],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99539477,0.0010959291,0.00025600212,0.001835595,0.0009412009,0.00047650025],"domain_scores_gemma":[0.9949432,0.0010109836,0.00016130586,0.0019327725,0.00141652,0.00053516566],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044729956,0.004955817,0.002932136,0.004445743,0.0032658982,0.0048177843,0.004670256,0.0031770335,0.06625327],"category_scores_gemma":[0.009551177,0.001671358,0.0025346002,0.0030588605,0.0015096609,0.005031053,0.0074263904,0.00349765,0.07936787],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009710314,0.00037485396,0.0012454503,0.0014718989,0.00018625283,0.0004802344,0.00042906016,0.0025128035,0.0128517095,0.004708814,0.74220705,0.23256098],"study_design_scores_gemma":[0.0011119843,0.00044569647,0.007388484,0.0007437372,0.0002050087,0.0015631167,0.0014479964,0.08015189,0.064139605,0.023692867,0.81871015,0.00039956122],"about_ca_topic_score_codex":0.06894369,"about_ca_topic_score_gemma":0.088563114,"teacher_disagreement_score":0.06894369,"about_ca_system_score_codex":0.0031138596,"about_ca_system_score_gemma":0.004675064,"threshold_uncertainty_score":0.22163928},"labels":[],"label_agreement":null},{"id":"W4385571036","doi":"10.18653/v1/2023.findings-acl.120","title":"Pragmatic Inference with a CLIP Listener for Contrastive Captioning","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Samsung","keywords":"Closed captioning; Discriminative model; Computer science; Leverage (statistics); Artificial intelligence; Inference; Fluency; Speech recognition; Natural language processing; Prosody; Hyperparameter; Image (mathematics); Linguistics","score_opus":0.018521158491553363,"score_gpt":0.31106428574775147,"score_spread":0.29254312725619813,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385571036","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0062312554,0.00019819145,0.9839094,0.00031241,0.00011860336,0.00014140553,0.00020351853,0.0044275983,0.004457548],"genre_scores_gemma":[0.3269249,0.00020089844,0.6624739,0.00096407434,0.00033742256,0.00040844723,0.0012635047,0.0013724505,0.0060544414],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9967554,0.0013773252,0.00011708435,0.0009082443,0.0006638397,0.0001781459],"domain_scores_gemma":[0.995191,0.0027893055,0.00028713734,0.0008517722,0.0006608975,0.0002199701],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003496771,0.0018686071,0.0009611255,0.0010267269,0.0009528279,0.0026036324,0.0026795913,0.002166436,0.010617816],"category_scores_gemma":[0.017940558,0.00085043686,0.0013430992,0.0005388044,0.0017834278,0.0038116612,0.0029895396,0.003723647,0.0032459528],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001017317,0.00037626966,0.002246405,0.0008979692,0.0003310924,0.0007655935,0.002302025,0.12762383,0.079121135,0.11065356,0.04038267,0.63428223],"study_design_scores_gemma":[0.000057713536,0.00011183335,0.00046185916,0.000041624404,0.00006237555,0.00021574479,0.00014344235,0.9143553,0.022327006,0.05049342,0.011661203,0.000068598645],"about_ca_topic_score_codex":0.0022891783,"about_ca_topic_score_gemma":0.003214849,"teacher_disagreement_score":0.010617816,"about_ca_system_score_codex":0.0014035807,"about_ca_system_score_gemma":0.0012105573,"threshold_uncertainty_score":0.035520136},"labels":[],"label_agreement":null},{"id":"W4385572364","doi":"10.18653/v1/2023.findings-acl.590","title":"Zero-shot Visual Question Answering with Language Model Feedback","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"National Natural Science Foundation of China","keywords":"Closed captioning; Computer science; Leverage (statistics); Language model; Task (project management); Artificial intelligence; Context (archaeology); Question answering; Natural language processing; Code (set theory); Scheme (mathematics); Machine learning; Speech recognition; Image (mathematics); Programming language; Engineering","score_opus":0.01670687766372517,"score_gpt":0.3167307179897402,"score_spread":0.300023840326015,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385572364","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016876644,0.001976442,0.9418402,0.00077880634,0.00038693292,0.00046989226,0.0013860037,0.030599823,0.005685203],"genre_scores_gemma":[0.40497082,0.0006877096,0.57021564,0.0021380698,0.00049425545,0.00066333433,0.009445169,0.001473857,0.009911025],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9970162,0.0012178043,0.000109391396,0.001004886,0.00047464782,0.00017710519],"domain_scores_gemma":[0.99510807,0.002790024,0.00021717045,0.0008185476,0.0008435056,0.00022252923],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024545148,0.0023673938,0.0015352088,0.0015603,0.00084773905,0.0019325964,0.004339433,0.0031606487,0.010510344],"category_scores_gemma":[0.011483181,0.00055690174,0.0012484185,0.00093236944,0.0012082803,0.0046598446,0.00313288,0.002706946,0.004046513],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00077593315,0.0007822505,0.0014033655,0.0015556857,0.0002059858,0.0004619479,0.0014044514,0.062534854,0.048023187,0.013884412,0.064825036,0.80414283],"study_design_scores_gemma":[0.00012842786,0.00041050967,0.0006708538,0.000099945915,0.000105338855,0.00033667375,0.00039622636,0.9099296,0.02714532,0.035742585,0.024929993,0.000104572835],"about_ca_topic_score_codex":0.0066227457,"about_ca_topic_score_gemma":0.007930041,"teacher_disagreement_score":0.010510344,"about_ca_system_score_codex":0.001515173,"about_ca_system_score_gemma":0.0014618023,"threshold_uncertainty_score":0.03516066},"labels":[],"label_agreement":null},{"id":"W4385572788","doi":"10.18653/v1/2022.conll-1.19","title":"Visual Semantic Parsing: From Images to Abstract Meaning Representation","year":2022,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Parsing; Computer science; Natural language processing; Meaning (existential); Representation (politics); Artificial intelligence; Bottom-up parsing; S-attributed grammar; Top-down parsing; Linguistics; Psychology; Philosophy","score_opus":0.02291593934030156,"score_gpt":0.3295676427037844,"score_spread":0.30665170336348285,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385572788","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012338805,0.0059756963,0.93081367,0.0035599,0.0007374117,0.00023888987,0.00549163,0.024794497,0.016049583],"genre_scores_gemma":[0.21982591,0.0059118634,0.74324775,0.0014637939,0.00040989724,0.00043233964,0.017417256,0.0027388157,0.008552362],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993672,0.00019684418,0.000050598042,0.00021929295,0.000098424855,0.00006771458],"domain_scores_gemma":[0.9988049,0.00048102092,0.00006691234,0.00037622167,0.00019755786,0.00007333815],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009805537,0.0012239038,0.0010330392,0.003245763,0.00073982246,0.004296182,0.0023928734,0.0017599888,0.011349963],"category_scores_gemma":[0.004305181,0.0010809994,0.0020959587,0.002685592,0.001820196,0.009810683,0.0035816825,0.002550495,0.004787175],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043208618,0.00016475467,0.0006569818,0.0007517353,0.00013261521,0.00033388048,0.00067714404,0.007026621,0.010467587,0.105115086,0.12679309,0.7474484],"study_design_scores_gemma":[0.00010685033,0.00011097969,0.0014358462,0.00037037107,0.0002018537,0.00031388202,0.0008326676,0.15524931,0.018253054,0.7335003,0.08953167,0.00009322344],"about_ca_topic_score_codex":0.0077902675,"about_ca_topic_score_gemma":0.006527622,"teacher_disagreement_score":0.011349963,"about_ca_system_score_codex":0.0009877076,"about_ca_system_score_gemma":0.0010982545,"threshold_uncertainty_score":0.03796947},"labels":[],"label_agreement":null},{"id":"W4385573082","doi":"10.18653/v1/2022.emnlp-main.92","title":"Generative Multi-hop Retrieval","year":2022,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Samsung; Ministry of Science and ICT, South Korea; Korea Advanced Institute of Science and Technology","keywords":"Computer science; Encoder; Vector space model; Embedding; Information retrieval; Bottleneck; Generative grammar; Artificial intelligence; Data mining; Theoretical computer science","score_opus":0.02634004319737722,"score_gpt":0.2973385893803788,"score_spread":0.2709985461830016,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385573082","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019046742,0.0015366008,0.97072554,0.00033557595,0.00009355124,0.00015541258,0.00037092218,0.0040183375,0.0037174462],"genre_scores_gemma":[0.5766457,0.001551092,0.3934672,0.0009045296,0.00029490667,0.00033809117,0.0024147958,0.000892402,0.023491265],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9987877,0.00030151507,0.00007993554,0.00031429276,0.00038938475,0.0001272197],"domain_scores_gemma":[0.9982376,0.0007944833,0.00010496916,0.0005320342,0.0002634275,0.00006742968],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011718264,0.0010253805,0.0016687376,0.0010549005,0.0005680472,0.0015409287,0.0027285186,0.0018541826,0.0049377484],"category_scores_gemma":[0.004695898,0.00059646834,0.0011357333,0.0015230748,0.000983018,0.0032059664,0.0023919686,0.0013855723,0.0033717074],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004949102,0.00031116136,0.0019311451,0.00065265247,0.00020070805,0.00074601633,0.00057555747,0.39374,0.0292542,0.04557342,0.019585278,0.506935],"study_design_scores_gemma":[0.000034290166,0.000093760864,0.00024859837,0.000015819747,0.000043137752,0.0005410828,0.000055856563,0.968473,0.009557914,0.017052772,0.0038406937,0.000043053456],"about_ca_topic_score_codex":0.00375414,"about_ca_topic_score_gemma":0.0052054604,"teacher_disagreement_score":0.0049377484,"about_ca_system_score_codex":0.00095869636,"about_ca_system_score_gemma":0.001068296,"threshold_uncertainty_score":0.016518354},"labels":[],"label_agreement":null},{"id":"W4385573711","doi":"10.18653/v1/2022.findings-emnlp.385","title":"Continuation KD: Improved Knowledge Distillation through the Lens of Continuation Optimization","year":2022,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Continuation; Computer science; Benchmark (surveying); Generalization; Distillation; Artificial intelligence; Limiting; Noise (video); Machine learning; Image (mathematics); Mathematics; Engineering; Programming language","score_opus":0.016878236703238153,"score_gpt":0.27409466346508604,"score_spread":0.25721642676184786,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385573711","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01040448,0.0003652095,0.9844431,0.00032957873,0.000065541586,0.000040265902,0.000080544145,0.0017703697,0.0025009336],"genre_scores_gemma":[0.388188,0.00044545226,0.6015686,0.0005997515,0.00014786626,0.00029582603,0.00058093516,0.0006958921,0.007477689],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9995889,0.00012655111,0.000022533886,0.000104751496,0.00010216345,0.00005512152],"domain_scores_gemma":[0.9987708,0.00067270617,0.00008014395,0.00022143315,0.00016473823,0.000090101385],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012720374,0.0011883843,0.0013808557,0.0007344923,0.00069372874,0.0011758768,0.0020640877,0.0019020793,0.004285342],"category_scores_gemma":[0.004817343,0.00066476146,0.0008366422,0.00074785115,0.0015976675,0.002547973,0.002599642,0.00294459,0.0014594294],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023275899,0.00013281852,0.0008098575,0.00021415684,0.00007953057,0.00016018924,0.00023048317,0.64210254,0.00707173,0.056417424,0.010254067,0.28229448],"study_design_scores_gemma":[0.0000134422335,0.000025465806,0.000036229674,0.000010637314,0.000004729435,0.00001540209,0.000008115217,0.98293144,0.0009216633,0.014932797,0.0010927563,0.0000074266086],"about_ca_topic_score_codex":0.0047903205,"about_ca_topic_score_gemma":0.005621854,"teacher_disagreement_score":0.0047903205,"about_ca_system_score_codex":0.000871633,"about_ca_system_score_gemma":0.0023948334,"threshold_uncertainty_score":0.014335871},"labels":[],"label_agreement":null},{"id":"W4385573873","doi":"10.18653/v1/2022.emnlp-main.586","title":"Contrastive Learning with Expectation-Maximization for Weakly Supervised Phrase Grounding","year":2022,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Fundamental Research Funds for the Central Universities; State Key Laboratory of Software Development Environment; Leverhulme Trust","keywords":"Phrase; Computer science; Margin (machine learning); Artificial intelligence; Annotation; Object (grammar); Natural language processing; Perspective (graphical); Expectation–maximization algorithm; Maximization; Pattern recognition (psychology); Speech recognition; Machine learning; Maximum likelihood; Mathematics","score_opus":0.008977920109588727,"score_gpt":0.2370852657185742,"score_spread":0.22810734560898546,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385573873","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011452718,0.0003665107,0.9844796,0.0002581355,0.000040439343,0.00007960902,0.00018662227,0.0022192718,0.00091714493],"genre_scores_gemma":[0.46006757,0.0004047572,0.52878654,0.0011810022,0.00031860507,0.0005412465,0.0025909194,0.0010183288,0.005091059],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9973707,0.0010988297,0.00010892905,0.00086724694,0.00036973308,0.00018447924],"domain_scores_gemma":[0.9945352,0.0036660666,0.00035369085,0.0006903448,0.00056810316,0.00018664006],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004219806,0.0024764591,0.002246141,0.0014405532,0.0006910425,0.0014285865,0.0041632266,0.0029254158,0.0035643105],"category_scores_gemma":[0.01176789,0.00093536044,0.0014987992,0.0014586117,0.0019314749,0.0034232833,0.0028360605,0.0039691846,0.0024139176],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009509607,0.0005599298,0.0026816234,0.0005716382,0.0003197609,0.00038432874,0.00033901175,0.45548198,0.019991823,0.019856397,0.01563417,0.48322845],"study_design_scores_gemma":[0.000030680367,0.0000819537,0.00019382325,0.00001262006,0.000016113067,0.00004193485,0.000016019785,0.9842741,0.002820978,0.011808955,0.0006891611,0.000013637162],"about_ca_topic_score_codex":0.0021763241,"about_ca_topic_score_gemma":0.0032571845,"teacher_disagreement_score":0.004219806,"about_ca_system_score_codex":0.0013401442,"about_ca_system_score_gemma":0.0012964723,"threshold_uncertainty_score":0.022316694},"labels":[],"label_agreement":null},{"id":"W4385574225","doi":"10.18653/v1/2022.findings-emnlp.31","title":"Lexicon-Enhanced Self-Supervised Training for Multilingual Dense Retrieval","year":2022,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Artificial intelligence; Relevance (law); Lexicon; Generator (circuit theory); Training set; Labeled data; Natural language processing; Machine learning; Information retrieval","score_opus":0.03598622992703411,"score_gpt":0.30980805225086244,"score_spread":0.27382182232382835,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385574225","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.092397384,0.0013445618,0.8826784,0.0003257326,0.00011491584,0.00034831336,0.0010641711,0.017214278,0.004512372],"genre_scores_gemma":[0.6265485,0.0004243368,0.35224226,0.0008131978,0.00018493706,0.0006037871,0.010380268,0.0008827541,0.007919778],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9988067,0.00043411506,0.00008804171,0.00036150683,0.0001960324,0.00011365205],"domain_scores_gemma":[0.99724305,0.001293449,0.00014974357,0.0005993778,0.00061950844,0.000094890784],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017999413,0.0012408983,0.0013873344,0.00156487,0.0006689076,0.00067402254,0.0023612056,0.0011488863,0.0031085548],"category_scores_gemma":[0.0057770936,0.0005972087,0.00090578606,0.0014987577,0.00084847846,0.0026595641,0.0016203807,0.0017745205,0.0028911023],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00056091807,0.0009486948,0.0044641146,0.0004191724,0.00018528833,0.00026701528,0.0003537678,0.109421276,0.027699329,0.0034460258,0.025557918,0.8266765],"study_design_scores_gemma":[0.00006554585,0.00015688641,0.0007183444,0.000016292093,0.000033448534,0.00016023434,0.00008439586,0.9820971,0.009451978,0.004075851,0.0031109685,0.000028948396],"about_ca_topic_score_codex":0.0071688173,"about_ca_topic_score_gemma":0.015297079,"teacher_disagreement_score":0.0071688173,"about_ca_system_score_codex":0.0007558067,"about_ca_system_score_gemma":0.0015295881,"threshold_uncertainty_score":0.014254212},"labels":[],"label_agreement":null},{"id":"W4385574233","doi":"10.18653/v1/2022.findings-emnlp.519","title":"Lexi: Self-Supervised Learning of the UI Language","year":2022,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Leverage (statistics); Natural language processing; Metadata; Artificial intelligence; Context (archaeology); Human–computer interaction; Task (project management); Information retrieval; World Wide Web","score_opus":0.006479628782001468,"score_gpt":0.2396702365762321,"score_spread":0.23319060779423062,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385574233","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08151632,0.0037887918,0.7853271,0.0015401277,0.0008190938,0.00083145965,0.013816404,0.10335187,0.009008872],"genre_scores_gemma":[0.452775,0.00086332636,0.43778074,0.0021962628,0.00036085973,0.0015417311,0.08808406,0.0025810222,0.013817023],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9983152,0.00041708956,0.00010121889,0.0007725497,0.00022874247,0.00016510936],"domain_scores_gemma":[0.9974521,0.0012668981,0.0001507208,0.00055696035,0.00042954218,0.00014376873],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024457416,0.003351798,0.0012085768,0.0021746857,0.0007990424,0.0017125411,0.004304028,0.002852453,0.0047738287],"category_scores_gemma":[0.007550367,0.00074120634,0.002128016,0.0016025114,0.0010708057,0.0026234214,0.0021494667,0.004573207,0.004547942],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058487064,0.0009934806,0.0062522474,0.000962364,0.00044133663,0.00043724006,0.0003350216,0.09239643,0.011726633,0.0048965295,0.17879787,0.702176],"study_design_scores_gemma":[0.00009088707,0.00018808403,0.0008822079,0.000061862265,0.000049734415,0.00013799894,0.00011359324,0.97885793,0.00654892,0.0064190533,0.0066155503,0.000034152963],"about_ca_topic_score_codex":0.007785581,"about_ca_topic_score_gemma":0.013571946,"teacher_disagreement_score":0.007785581,"about_ca_system_score_codex":0.0019072146,"about_ca_system_score_gemma":0.0019185425,"threshold_uncertainty_score":0.015970051},"labels":[],"label_agreement":null},{"id":"W4385612782","doi":"10.1145/3539618.3591903","title":"AToMiC: An Image/Text Retrieval Test Collection to Support Multimedia Content Creation","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Test (biology); Multimedia; Content (measure theory); Information retrieval; Image retrieval; Content-based image retrieval; Image (mathematics); Artificial intelligence","score_opus":0.037569380739278235,"score_gpt":0.3233776406108255,"score_spread":0.2858082598715473,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385612782","genre_codex":"dataset","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14117639,0.004065787,0.057489328,0.0017255411,0.0018850726,0.005970068,0.6890669,0.0740936,0.024527377],"genre_scores_gemma":[0.048238687,0.00039030952,0.053125367,0.00046654986,0.000223198,0.0015451784,0.8878052,0.0015173439,0.006688157],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99705446,0.0006231572,0.0003398018,0.0006690689,0.0010313846,0.00028211277],"domain_scores_gemma":[0.9940627,0.001368796,0.000336648,0.0020292145,0.0015038962,0.0006986707],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027238063,0.0027278038,0.0013482004,0.0062688063,0.0016926714,0.0021793677,0.0038755215,0.0026938643,0.01209863],"category_scores_gemma":[0.009756105,0.00061316,0.0017516596,0.0044874875,0.0012252821,0.0036618481,0.0035784063,0.0023065556,0.018199932],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010918424,0.0017739127,0.006210141,0.0025861883,0.000309649,0.00062651536,0.00041577208,0.005611339,0.015880143,0.0023155054,0.82746553,0.13571347],"study_design_scores_gemma":[0.0017485169,0.0024822222,0.03730281,0.0005033955,0.0004894325,0.0044024964,0.0019840456,0.1402342,0.07757399,0.00688846,0.72582036,0.00057009497],"about_ca_topic_score_codex":0.015360942,"about_ca_topic_score_gemma":0.03332415,"teacher_disagreement_score":0.015360942,"about_ca_system_score_codex":0.0017018603,"about_ca_system_score_gemma":0.0018252713,"threshold_uncertainty_score":0.040473998},"labels":[],"label_agreement":null},{"id":"W4385784741","doi":"10.1007/978-3-319-08234-9_501-1","title":"Automated Image Captioning for the Visually Impaired","year":2023,"lang":"en","type":"book-chapter","venue":"Encyclopedia of Computer Graphics and Games","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"CNIB Foundation; Ontario Tech University","funders":"","keywords":"Closed captioning; Visually impaired; Computer science; Computer vision; Image (mathematics); Artificial intelligence; Human–computer interaction","score_opus":0.01395764660038844,"score_gpt":0.2692701808536056,"score_spread":0.25531253425321715,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385784741","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08131882,0.014033467,0.7251344,0.0024835682,0.002537484,0.00078621454,0.0051080524,0.03287939,0.13571872],"genre_scores_gemma":[0.26396224,0.014260104,0.55470026,0.0006992457,0.00085822743,0.00048444484,0.009348892,0.0030636273,0.15262295],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99983025,0.000040207302,0.000010941236,0.000031301057,0.000067408444,0.000019818179],"domain_scores_gemma":[0.9995597,0.00015560671,0.000020457715,0.00006455968,0.00017612417,0.000023544955],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00029549445,0.0008719172,0.0003972413,0.0010812255,0.00032382648,0.0014470498,0.00091908197,0.0009274959,0.03189453],"category_scores_gemma":[0.0017340957,0.00024153381,0.0003773637,0.0005785095,0.00034818577,0.0010177662,0.0008715833,0.00059823756,0.009896162],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023450605,0.0000747214,0.000214164,0.0003843173,0.000014259087,0.00037435527,0.00015876147,0.0033757058,0.024986003,0.0021498573,0.1003491,0.86768425],"study_design_scores_gemma":[0.00011593609,0.00077688566,0.008853841,0.0009820034,0.00018333462,0.007418141,0.0013401827,0.2723183,0.15851387,0.025130823,0.5241697,0.0001970126],"about_ca_topic_score_codex":0.0024179504,"about_ca_topic_score_gemma":0.002279569,"teacher_disagreement_score":0.03189453,"about_ca_system_score_codex":0.00026523863,"about_ca_system_score_gemma":0.0003985239,"threshold_uncertainty_score":0.10669786},"labels":[],"label_agreement":null},{"id":"W4385832667","doi":"10.1007/978-3-031-39821-6_7","title":"Efficient Video Captioning with Frame Similarity-Based Filtering","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Closed captioning; Computer science; Artificial intelligence; Frame (networking); Similarity (geometry); Task (project management); Feature (linguistics); Convolutional neural network; Convolution (computer science); Computer vision; Pattern recognition (psychology); Speech recognition; Artificial neural network; Natural language processing; Image (mathematics)","score_opus":0.017219776979267562,"score_gpt":0.2568358270065685,"score_spread":0.2396160500273009,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385832667","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0066147847,0.0009193467,0.9845095,0.00009365958,0.00026497923,0.00014930933,0.0003382465,0.004280647,0.002829575],"genre_scores_gemma":[0.064526424,0.0012862558,0.92172074,0.00014883187,0.00036506003,0.00018832853,0.0023783676,0.0008329319,0.008553027],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991442,0.00011876676,0.000048691974,0.00017982465,0.00040458154,0.00010384184],"domain_scores_gemma":[0.9989662,0.0002895026,0.00005590816,0.00019531758,0.00043765985,0.00005553141],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00068111235,0.0018959119,0.002003502,0.003133684,0.00067409,0.0018615571,0.0015901385,0.0013772314,0.0114855245],"category_scores_gemma":[0.0024457008,0.0006151139,0.0013640176,0.0031868704,0.00042437963,0.0017316319,0.001271697,0.0013372946,0.008118578],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044220244,0.00011829771,0.00016694856,0.00021677812,0.000055104436,0.00011703964,0.00004436896,0.008951407,0.093264885,0.0024861237,0.01266802,0.8814689],"study_design_scores_gemma":[0.00004566961,0.00021240195,0.0009877086,0.00004106204,0.00011217792,0.00054702576,0.000067447436,0.83294827,0.14444992,0.004010279,0.016527312,0.000050667826],"about_ca_topic_score_codex":0.004569843,"about_ca_topic_score_gemma":0.0048341355,"teacher_disagreement_score":0.0114855245,"about_ca_system_score_codex":0.00059260626,"about_ca_system_score_gemma":0.0006911607,"threshold_uncertainty_score":0.038422883},"labels":[],"label_agreement":null},{"id":"W4386065347","doi":"10.1109/cvpr52729.2023.00646","title":"Meta-Explore: Exploratory Hierarchical Vision-and-Language Navigation Using Scene Object Spectrum Grounding","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Benchmark (surveying); Object (grammar); Generalization; Representation (politics); Natural language; Artificial intelligence; Action (physics); Domain (mathematical analysis); Path (computing); Semantics (computer science); State (computer science); Human–computer interaction; Programming language","score_opus":0.07991573117494452,"score_gpt":0.34829690584924206,"score_spread":0.26838117467429756,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386065347","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.053279057,0.00081452937,0.9149097,0.0003106943,0.000082186874,0.00015198354,0.00058217585,0.025594898,0.004274782],"genre_scores_gemma":[0.4277713,0.0002841678,0.5637682,0.00043136466,0.000035298115,0.00027231345,0.0020589472,0.0011152946,0.0042631],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996773,0.000059856367,0.000011273439,0.00011785614,0.0000831273,0.0000504626],"domain_scores_gemma":[0.9996369,0.00013854472,0.000034013065,0.0001005374,0.00005433184,0.000035746154],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005180326,0.0014662399,0.0009837389,0.00075842603,0.00049502915,0.000846567,0.002291968,0.0012954385,0.0025580125],"category_scores_gemma":[0.0013619849,0.0005286265,0.0011609476,0.0005612152,0.0007022233,0.0017036467,0.0021022544,0.0013375711,0.0008324165],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004730405,0.00032437718,0.0030837941,0.00032485116,0.00017989376,0.00041516894,0.00055451225,0.48617673,0.027787188,0.01556982,0.017577516,0.44753304],"study_design_scores_gemma":[0.000040396204,0.00008389075,0.00022973938,0.000013977074,0.000015433572,0.0000612981,0.000055256045,0.98564523,0.0035691196,0.007617045,0.0026511562,0.000017440778],"about_ca_topic_score_codex":0.01175584,"about_ca_topic_score_gemma":0.019977033,"teacher_disagreement_score":0.01175584,"about_ca_system_score_codex":0.00074563595,"about_ca_system_score_gemma":0.0014309325,"threshold_uncertainty_score":0.023374856},"labels":[],"label_agreement":null},{"id":"W4386071509","doi":"10.1109/cvpr52729.2023.00690","title":"PACO: Parts and Attributes of Common Objects","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":62,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Object (grammar); Benchmark (surveying); Artificial intelligence; Code (set theory); Segmentation; Object detection; Information retrieval; Natural language processing; Computer vision; Pattern recognition (psychology); Set (abstract data type); Programming language; Geography","score_opus":0.021330160297669664,"score_gpt":0.2892003363021184,"score_spread":0.26787017600444873,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386071509","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.105881974,0.01042155,0.4084335,0.001706262,0.0010229178,0.0011921737,0.33584067,0.113965906,0.021535084],"genre_scores_gemma":[0.12395159,0.0017991333,0.26401567,0.0007680674,0.00017006135,0.0008097706,0.59880996,0.0036266334,0.006049213],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99708027,0.00026210854,0.00017140355,0.0014752523,0.000784176,0.00022677524],"domain_scores_gemma":[0.9960311,0.000887566,0.0002639009,0.002035164,0.0005217973,0.0002603691],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018455208,0.003287708,0.0019148755,0.004635334,0.0013094747,0.0036867561,0.005657102,0.0027608988,0.0064429482],"category_scores_gemma":[0.008185854,0.0013644242,0.0027882035,0.003839849,0.0013203928,0.0080808895,0.0055326563,0.0030857949,0.007811072],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020673678,0.0005114728,0.031520996,0.0037642296,0.00064428203,0.000541787,0.00068890775,0.029627275,0.023650795,0.018700913,0.41313112,0.47515085],"study_design_scores_gemma":[0.00019074159,0.0005868999,0.041473467,0.0009296761,0.0003962459,0.0027203616,0.0008504705,0.35842472,0.046722095,0.05592572,0.49147803,0.0003015259],"about_ca_topic_score_codex":0.01734204,"about_ca_topic_score_gemma":0.031574924,"teacher_disagreement_score":0.01734204,"about_ca_system_score_codex":0.0020176903,"about_ca_system_score_gemma":0.0019013262,"threshold_uncertainty_score":0.03448218},"labels":[],"label_agreement":null},{"id":"W4386076676","doi":"10.1109/cvpr52729.2023.00246","title":"Make-A-Story: Visual Memory Conditioned Consistent Story Generation","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":42,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; Canadian Institute for Advanced Research; University of British Columbia","funders":"","keywords":"Computer science; Consistency (knowledge bases); Generative grammar; Context (archaeology); Artificial intelligence; Task (project management); Sentence; Natural language processing; Visualization; Quality (philosophy); Generative model","score_opus":0.03342713523481007,"score_gpt":0.3042961115903501,"score_spread":0.27086897635554,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386076676","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06603357,0.0026180497,0.8899651,0.0007606479,0.0002915065,0.0003393206,0.0033623814,0.028100744,0.008528688],"genre_scores_gemma":[0.48950848,0.0006646045,0.48316357,0.00053929875,0.00015944225,0.00027841795,0.01164145,0.0016966915,0.01234798],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995086,0.00014913183,0.000021606817,0.0001900991,0.000085808024,0.00004483467],"domain_scores_gemma":[0.99908507,0.0004295595,0.000064165,0.00025460124,0.0001004498,0.000066118155],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000756877,0.0015094825,0.00059048046,0.0008759546,0.00038495008,0.0009089022,0.0022173696,0.0012839749,0.0063607553],"category_scores_gemma":[0.0033791552,0.00047064334,0.0010007428,0.0004464345,0.00046551425,0.0018121713,0.0013637813,0.0015063515,0.0018740133],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00084473373,0.00042597295,0.0027902562,0.0005594638,0.00028196556,0.00087225187,0.00093230506,0.09080012,0.040130995,0.01576184,0.053919252,0.79268086],"study_design_scores_gemma":[0.00015498778,0.00026989338,0.0013294682,0.000047063124,0.00008027328,0.00048592855,0.00017470804,0.9371359,0.022672784,0.017855708,0.019744694,0.000048559345],"about_ca_topic_score_codex":0.0044283653,"about_ca_topic_score_gemma":0.010776996,"teacher_disagreement_score":0.0063607553,"about_ca_system_score_codex":0.00063396926,"about_ca_system_score_gemma":0.00050265185,"threshold_uncertainty_score":0.021278918},"labels":[],"label_agreement":null},{"id":"W4386159122","doi":"10.1109/icme55011.2023.00259","title":"CHAN: Cross-Modal Hybrid Attention Network for Temporal Language Grounding in Videos","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"China Postdoctoral Science Foundation","keywords":"Computer science; Modality (human–computer interaction); Modalities; Modal; Semantics (computer science); Sentence; Frame (networking); Key (lock); Artificial intelligence; Natural language processing; Word (group theory); Task (project management); Focus (optics); Speech recognition; Linguistics; Engineering","score_opus":0.02041856827068133,"score_gpt":0.34065249028803046,"score_spread":0.3202339220173491,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386159122","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.077341236,0.0029060622,0.90160775,0.0005872579,0.00027484016,0.00029641902,0.0014048156,0.00902285,0.006558872],"genre_scores_gemma":[0.8169597,0.00096546777,0.16510764,0.00091216544,0.00025739265,0.0003334274,0.0037122325,0.00037331163,0.011378644],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9995592,0.000084199644,0.000014739798,0.00017691327,0.00007406446,0.00009079883],"domain_scores_gemma":[0.99960905,0.00017064351,0.000038224956,0.000044027125,0.00010213551,0.000035915116],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007943106,0.0015576577,0.00082554895,0.0014894223,0.0005213437,0.000602668,0.0017951014,0.0010788902,0.0030938704],"category_scores_gemma":[0.0019963442,0.00040219873,0.00079460884,0.0011219853,0.0005073521,0.0017751423,0.0016109194,0.0011443904,0.000692189],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00085004454,0.00036947938,0.0030355398,0.00028619036,0.0002913558,0.00044971152,0.00032466534,0.13731892,0.03959001,0.007030446,0.020823443,0.7896302],"study_design_scores_gemma":[0.000025632578,0.00010082882,0.000947397,0.000017606448,0.00005669637,0.0000778535,0.00006074517,0.9842067,0.005922065,0.006241498,0.0023231339,0.000019848883],"about_ca_topic_score_codex":0.02596678,"about_ca_topic_score_gemma":0.032964956,"teacher_disagreement_score":0.02596678,"about_ca_system_score_codex":0.0013337738,"about_ca_system_score_gemma":0.000988748,"threshold_uncertainty_score":0.05163127},"labels":[],"label_agreement":null},{"id":"W4386162736","doi":"10.1145/3617592","title":"Deep Learning Approaches on Image Captioning: A Review","year":2023,"lang":"en","type":"review","venue":"ACM Computing Surveys","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":164,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Closed captioning; Computer science; Deep learning; Artificial intelligence; Modalities; Context (archaeology); Field (mathematics); Natural language processing; Image (mathematics); Machine learning","score_opus":0.17702160835122344,"score_gpt":0.3854094199385489,"score_spread":0.20838781158732544,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386162736","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0006307851,0.9811365,0.012340477,0.00094457506,0.00042983858,0.000040515897,0.00010981647,0.00014730832,0.0042203283],"genre_scores_gemma":[0.004821733,0.98393583,0.008288825,0.00047544163,0.00046653041,0.000044627395,0.00026831063,0.000045593515,0.0016531361],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99949634,0.00009883143,0.000059893206,0.00010498752,0.00020398588,0.000035957535],"domain_scores_gemma":[0.99702746,0.0019337973,0.00013070863,0.000103012724,0.0007372604,0.000067789086],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017438161,0.0014171947,0.0011386448,0.0032278632,0.0003329866,0.0015934865,0.001991772,0.001425617,0.005571042],"category_scores_gemma":[0.0051273326,0.0006834535,0.0008163552,0.0039150636,0.0006927071,0.0031537737,0.0011426738,0.0018000553,0.003517564],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000034610286,0.00005954921,0.00024980583,0.007723321,0.000070824964,0.0000519392,0.000065738124,0.0018603754,0.0006170807,0.0052787573,0.025916161,0.9580719],"study_design_scores_gemma":[0.000027504295,0.00022411188,0.0018707273,0.011503771,0.00030225728,0.0011663048,0.00019188947,0.008450468,0.0033177782,0.013437662,0.9594078,0.00009971162],"about_ca_topic_score_codex":0.0028668016,"about_ca_topic_score_gemma":0.0029743593,"teacher_disagreement_score":0.005571042,"about_ca_system_score_codex":0.0009490824,"about_ca_system_score_gemma":0.001456081,"threshold_uncertainty_score":0.018637002},"labels":[],"label_agreement":null},{"id":"W4386243186","doi":"10.1109/crv60082.2023.00048","title":"Few-Shot Personality-Specific Image Captioning via Meta-Learning","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; University of Manitoba","funders":"","keywords":"Closed captioning; Computer science; Focus (optics); Benchmark (surveying); Shot (pellet); Set (abstract data type); Artificial intelligence; Big Five personality traits; Machine learning; Image (mathematics); Personality; Information retrieval","score_opus":0.06926265009916809,"score_gpt":0.3147886427505567,"score_spread":0.24552599265138858,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386243186","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036656756,0.0022433756,0.93522865,0.0007605874,0.0006007909,0.0005247775,0.0014093848,0.016724253,0.0058513605],"genre_scores_gemma":[0.41091743,0.0012742382,0.56289583,0.0012411634,0.00057859416,0.000657393,0.0061453884,0.0012320349,0.015057961],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99911517,0.0002813894,0.00003099296,0.00037206896,0.0001246666,0.00007572128],"domain_scores_gemma":[0.99720937,0.0013779826,0.00017217253,0.000655404,0.00044000105,0.00014514137],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016221644,0.0018269054,0.0010815638,0.0010751688,0.00047444593,0.0013129809,0.0022477242,0.0020535784,0.004188679],"category_scores_gemma":[0.0071777515,0.0006173004,0.0011068553,0.0009047014,0.0006690029,0.0023898226,0.0012997858,0.002699313,0.0031067687],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058836857,0.0006187327,0.0020596127,0.00079064205,0.00023709753,0.00050955516,0.0004971661,0.15034248,0.03787784,0.0040946286,0.045538764,0.7568451],"study_design_scores_gemma":[0.000019704808,0.00015807287,0.0005747857,0.000040931147,0.000036734767,0.00026039724,0.00006569159,0.9740976,0.0145543795,0.005427802,0.0047279187,0.000036086178],"about_ca_topic_score_codex":0.002079954,"about_ca_topic_score_gemma":0.004367735,"teacher_disagreement_score":0.004188679,"about_ca_system_score_codex":0.0010314044,"about_ca_system_score_gemma":0.0006195289,"threshold_uncertainty_score":0.014012575},"labels":[],"label_agreement":null},{"id":"W4386249615","doi":"10.1109/crv60082.2023.00034","title":"Naive Scene Graphs: How Visual is Modern Visual Relationship Detection?","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Artificial intelligence; Scene graph; Categorical variable; Naive Bayes classifier; Pixel; Bounding overwatch; Pattern recognition (psychology); Classifier (UML); Object detection; Graph; Computer vision; Machine learning; Theoretical computer science; Support vector machine","score_opus":0.02509051419035251,"score_gpt":0.3185390179645395,"score_spread":0.29344850377418696,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386249615","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03608818,0.0012496762,0.94523007,0.002011843,0.00022069749,0.00019315892,0.000579697,0.008532961,0.0058937175],"genre_scores_gemma":[0.3611834,0.0007007952,0.63020355,0.0010416675,0.00017409104,0.00012691766,0.0022075782,0.00079426105,0.003567723],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99763846,0.0008920604,0.000064617125,0.0005885937,0.00068897655,0.00012735563],"domain_scores_gemma":[0.9954716,0.0020356681,0.00032703686,0.0010940599,0.00086912926,0.00020248246],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033739307,0.0010156928,0.00084571145,0.0022781033,0.0006974669,0.0024320844,0.0025923771,0.0014873899,0.0037324526],"category_scores_gemma":[0.011815064,0.00057544344,0.0008594246,0.0011342888,0.0011316491,0.0051871724,0.0013814738,0.0017039056,0.0027758875],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029371155,0.00020329948,0.010097347,0.00035784036,0.00013442145,0.00019177978,0.0003819591,0.029686792,0.0129970545,0.04077843,0.026702153,0.8781752],"study_design_scores_gemma":[0.00006466811,0.0002860017,0.006180563,0.00017621719,0.00007374576,0.00088628905,0.00056404,0.74203205,0.019026358,0.20267689,0.027942529,0.00009055515],"about_ca_topic_score_codex":0.0036758077,"about_ca_topic_score_gemma":0.006606424,"teacher_disagreement_score":0.0037324526,"about_ca_system_score_codex":0.000899347,"about_ca_system_score_gemma":0.00069332327,"threshold_uncertainty_score":0.017843306},"labels":[],"label_agreement":null},{"id":"W4386566765","doi":"10.18653/v1/2023.eacl-main.185","title":"MAPL: Parameter-Efficient Adaptation of Unimodal Pre-Trained Models for Vision-Language Few-Shot Prompting","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Canadian Institute for Advanced Research","funders":"Samsung; Canadian Institute for Advanced Research","keywords":"Computer science; Adaptation (eye); Artificial intelligence; Computational linguistics; Association (psychology); Shot (pellet); Natural language processing; Speech recognition; Psychology; Neuroscience; Chemistry","score_opus":0.043569357639590814,"score_gpt":0.3403577905972081,"score_spread":0.2967884329576173,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386566765","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014777119,0.0009815102,0.87980545,0.00019805212,0.0005290943,0.00019559807,0.0010484533,0.099253476,0.003211135],"genre_scores_gemma":[0.33056688,0.0006488507,0.63838905,0.00071835064,0.00024505812,0.0007566141,0.006255395,0.0070686266,0.015351068],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99947757,0.0001136759,0.000022241098,0.0002047941,0.00011035408,0.000071369366],"domain_scores_gemma":[0.99897087,0.00045793448,0.00003338502,0.00020340769,0.00023113187,0.00010333289],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010649266,0.0017302354,0.0014303358,0.0007305322,0.00044911847,0.0012500755,0.003681131,0.001793715,0.015320261],"category_scores_gemma":[0.0047015464,0.0007567443,0.0007753031,0.0005122645,0.0003890273,0.0020699794,0.003014379,0.0028753893,0.0085775135],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011875251,0.00044330917,0.00060165056,0.00035174165,0.000160239,0.00037606375,0.0002142718,0.077713,0.032256413,0.0029691851,0.05420884,0.8295178],"study_design_scores_gemma":[0.000061319195,0.00010365783,0.00030136458,0.000022800286,0.000023838935,0.00010050417,0.000060248323,0.97417575,0.014875173,0.004261294,0.005982837,0.000031161828],"about_ca_topic_score_codex":0.0077684782,"about_ca_topic_score_gemma":0.0118658105,"teacher_disagreement_score":0.015320261,"about_ca_system_score_codex":0.00080217724,"about_ca_system_score_gemma":0.0010376702,"threshold_uncertainty_score":0.05125141},"labels":[],"label_agreement":null},{"id":"W4386978004","doi":"10.48550/arxiv.2309.12314","title":"TinyCLIP: CLIP Distillation via Affinity Mimicking and Weight Inheritance","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"St. Stephen's University","funders":"National Natural Science Foundation of China","keywords":"Distillation; Computer science; Inheritance (genetic algorithm); Modal; Artificial intelligence; Feature (linguistics); AKA; Machine learning; Transferability; Code (set theory); Scratch; Image (mathematics); Pattern recognition (psychology); Programming language; Chemistry; Chromatography","score_opus":0.08317322424467298,"score_gpt":0.21212730965512777,"score_spread":0.12895408541045478,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386978004","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029132493,0.0005925686,0.92890227,0.00043156906,0.00033248818,0.00018507315,0.0009078726,0.033955023,0.0055605564],"genre_scores_gemma":[0.39438152,0.00037744245,0.57712424,0.0014608079,0.00020711047,0.00059221796,0.0049116546,0.0042521385,0.01669281],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99949896,0.00007548614,0.000021190233,0.00016330885,0.00015036714,0.000090663576],"domain_scores_gemma":[0.9994636,0.0001726847,0.00003448563,0.0001707722,0.00010737392,0.000051155297],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007016573,0.001814114,0.00083115784,0.0005008419,0.00063306023,0.0011619516,0.0031944425,0.0011827337,0.010774713],"category_scores_gemma":[0.0035547854,0.00067603966,0.0009270462,0.0006088504,0.00085061474,0.0027175082,0.003203025,0.0028836313,0.0035021636],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007432135,0.0002615246,0.0016271822,0.00041472737,0.00019776128,0.0004276303,0.00029291655,0.18760726,0.052588508,0.01711513,0.0338735,0.7048507],"study_design_scores_gemma":[0.00004989596,0.000099505065,0.00022617257,0.000021502,0.000024444604,0.00011295851,0.00004421823,0.9579373,0.02590538,0.0072720503,0.008267737,0.000038737613],"about_ca_topic_score_codex":0.006938132,"about_ca_topic_score_gemma":0.012595725,"teacher_disagreement_score":0.010774713,"about_ca_system_score_codex":0.0007537104,"about_ca_system_score_gemma":0.0014046329,"threshold_uncertainty_score":0.036044955},"labels":[],"label_agreement":null},{"id":"W4387076386","doi":"10.48550/arxiv.2309.14181","title":"Q-Bench: A Benchmark for General-Purpose Foundation Models on Low-level Vision","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Institute for Catastrophic Loss Reduction","keywords":"Perception; Computer science; Benchmark (surveying); Construct (python library); Correctness; Artificial intelligence; Softmax function; Pipeline (software); Quality (philosophy); Foundation (evidence); Visual perception; Natural language processing; Machine learning; Psychology; Deep learning; Programming language","score_opus":0.1809078712691692,"score_gpt":0.25629417249861564,"score_spread":0.07538630122944642,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387076386","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18262681,0.01840193,0.5645807,0.003091564,0.0015707036,0.0026945437,0.06731159,0.12726119,0.032461],"genre_scores_gemma":[0.4112552,0.0022157016,0.43992382,0.0020354781,0.00025008924,0.0017318766,0.13148062,0.0036750224,0.0074322466],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.995248,0.0015367796,0.00042103283,0.0014722998,0.0009984673,0.0003233845],"domain_scores_gemma":[0.9904146,0.005431821,0.0004711899,0.0018276152,0.0015053782,0.0003493812],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068972795,0.003603692,0.0013481202,0.0027644772,0.00087350165,0.0035651997,0.005766296,0.00430525,0.008180416],"category_scores_gemma":[0.029493423,0.00081177073,0.0026366662,0.0019182146,0.0012248782,0.004557303,0.0037207452,0.0033236677,0.0054428773],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018135227,0.0011176049,0.008505652,0.004832817,0.001305621,0.00058431533,0.00038049728,0.23343481,0.0155937625,0.009558611,0.14061365,0.58225924],"study_design_scores_gemma":[0.00025880127,0.00065058476,0.0042665135,0.00041756599,0.00014534585,0.0004660658,0.00024995668,0.94542664,0.014939089,0.013223862,0.019846201,0.00010936474],"about_ca_topic_score_codex":0.016693186,"about_ca_topic_score_gemma":0.018700846,"teacher_disagreement_score":0.016693186,"about_ca_system_score_codex":0.0027746612,"about_ca_system_score_gemma":0.0020743043,"threshold_uncertainty_score":0.03647679},"labels":[],"label_agreement":null},{"id":"W4387389875","doi":"10.48550/arxiv.2310.02567","title":"Improving Automatic VQA Evaluation Using Large Language Models","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Canadian Institute for Advanced Research; Samsung; Nvidia","keywords":"Leverage (statistics); Computer science; Metric (unit); Machine learning; Proxy (statistics); Question answering; Task (project management); Artificial intelligence; Set (abstract data type); Context (archaeology); Data mining; Information retrieval","score_opus":0.13621283737109485,"score_gpt":0.26115521070933334,"score_spread":0.12494237333823849,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387389875","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2393579,0.0064820317,0.6380796,0.002641955,0.001298874,0.0011783617,0.005204452,0.089386754,0.016370025],"genre_scores_gemma":[0.7915302,0.00039046584,0.18849337,0.0010561929,0.00023568963,0.0005712751,0.010491352,0.0026270284,0.0046043894],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.96706486,0.021750066,0.0018841252,0.0032141663,0.005335288,0.00075151667],"domain_scores_gemma":[0.934461,0.041512415,0.0019289961,0.006909536,0.014030688,0.0011573519],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01851711,0.0020511847,0.0015010936,0.0029237263,0.0011162683,0.0041216365,0.0027579134,0.0023307975,0.0068319514],"category_scores_gemma":[0.085687466,0.0006313966,0.0015076797,0.0012710836,0.0009446856,0.005575601,0.0041591856,0.0028456566,0.0037446981],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016605597,0.0010624521,0.018497484,0.0017365994,0.00056488544,0.00039498118,0.0024000306,0.09604312,0.04148013,0.011827359,0.09356888,0.7307635],"study_design_scores_gemma":[0.00025669188,0.0005429888,0.005034442,0.00016418836,0.00010153121,0.00022768724,0.0005850379,0.93820655,0.023743607,0.012930714,0.018059717,0.00014688206],"about_ca_topic_score_codex":0.009747172,"about_ca_topic_score_gemma":0.01153384,"teacher_disagreement_score":0.01851711,"about_ca_system_score_codex":0.002185546,"about_ca_system_score_gemma":0.0023178481,"threshold_uncertainty_score":0.097929},"labels":[],"label_agreement":null},{"id":"W4387969760","doi":"10.1145/3581783.3612863","title":"Deep Video Understanding with Video-Language Model","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"National Natural Science Foundation of China; Fundamental Research Funds for the Central Universities; National Science Foundation","keywords":"Computer science; Process (computing); Artificial intelligence; Selection (genetic algorithm); Matching (statistics); Modal; Graph; Dual (grammatical number); Multimedia; Machine learning; Human–computer interaction; Natural language processing; Theoretical computer science; Programming language","score_opus":0.035195777888689314,"score_gpt":0.2896068538903968,"score_spread":0.2544110760017075,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387969760","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03715958,0.0009220447,0.9449278,0.0006040392,0.00010509158,0.00013580096,0.0011493814,0.011867418,0.0031286902],"genre_scores_gemma":[0.6587274,0.00065250025,0.32290843,0.0007566627,0.000112914146,0.00025800423,0.006394503,0.0005194838,0.009670133],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99968684,0.00005315152,0.000012659894,0.00015426835,0.000050325347,0.000042884803],"domain_scores_gemma":[0.9996196,0.00015819963,0.000038940572,0.000071540206,0.00008422537,0.00002739341],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00043550623,0.0012847749,0.0005635819,0.00078843714,0.00026121,0.0009166669,0.001629865,0.0012274279,0.0029287976],"category_scores_gemma":[0.0018915613,0.0003182816,0.0010616624,0.0006924612,0.00031645107,0.0022792101,0.0009170928,0.0019739857,0.0015727174],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020701486,0.00029922562,0.0013987803,0.00019559832,0.00016293424,0.00025850243,0.00019644141,0.34918398,0.023283293,0.009696514,0.017834056,0.5972836],"study_design_scores_gemma":[0.00000583078,0.00003178403,0.00012889781,0.000007346831,0.000012750155,0.000026618422,0.00002629045,0.99071735,0.0029846635,0.0048926985,0.0011596313,0.000006140433],"about_ca_topic_score_codex":0.01443951,"about_ca_topic_score_gemma":0.017262973,"teacher_disagreement_score":0.01443951,"about_ca_system_score_codex":0.0010206107,"about_ca_system_score_gemma":0.0008627241,"threshold_uncertainty_score":0.028710961},"labels":[],"label_agreement":null},{"id":"W4388937551","doi":"10.1109/icccnt56998.2023.10307136","title":"Assistive Application for the Visually Impaired using Machine Learning and Image Processing","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Horizon College and Seminary","funders":"","keywords":"Closed captioning; Computer science; Artificial intelligence; Benchmark (surveying); Metric (unit); Task (project management); Process (computing); Visually impaired; DECIPHER; Visualization; Image (mathematics); Image processing; Computer vision; Natural language processing; Machine learning; Pattern recognition (psychology); Human–computer interaction","score_opus":0.02367214711049958,"score_gpt":0.34182813294411823,"score_spread":0.31815598583361865,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388937551","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19731924,0.0056619598,0.75361276,0.0013707576,0.00052148005,0.0006136344,0.00251646,0.016406659,0.021977073],"genre_scores_gemma":[0.67032814,0.0021968973,0.31010994,0.00055783737,0.00010769479,0.00030124153,0.0018115027,0.00016373627,0.014423033],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9998085,0.00005130278,0.000014743797,0.000040966905,0.000062025836,0.00002248997],"domain_scores_gemma":[0.9996051,0.00016431294,0.000028213712,0.00006776927,0.000115530675,0.000018947769],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00037358384,0.0007445329,0.00041548448,0.0006388181,0.00032639562,0.0007047373,0.0006517044,0.0007563395,0.0047370815],"category_scores_gemma":[0.0015445137,0.00010888996,0.00045748337,0.0004098254,0.00027791117,0.00097232853,0.0006883112,0.00047112518,0.0018179726],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034952076,0.0002769629,0.002449415,0.00080476794,0.000072000534,0.0011527945,0.0003357667,0.015127446,0.053912602,0.0017736335,0.018912734,0.9048324],"study_design_scores_gemma":[0.000071583876,0.00092984084,0.0121878125,0.00045900332,0.0002410398,0.004264607,0.0011085505,0.73802555,0.159955,0.016114777,0.066460274,0.00018192905],"about_ca_topic_score_codex":0.0032693087,"about_ca_topic_score_gemma":0.0047330502,"teacher_disagreement_score":0.0047370815,"about_ca_system_score_codex":0.00037167265,"about_ca_system_score_gemma":0.00040940312,"threshold_uncertainty_score":0.015847087},"labels":[],"label_agreement":null},{"id":"W4389285856","doi":"10.1007/978-3-031-48796-5_10","title":"StableYolo: Optimizing Image Generation for Large Language Models","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Image (mathematics); Inference; Process (computing); Artificial intelligence; Computation; Bounded function; Approx; Baseline (sea); Image quality; Computer vision; Algorithm; Programming language; Operating system; Mathematics","score_opus":0.02895945149165552,"score_gpt":0.2932319525451203,"score_spread":0.2642725010534648,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389285856","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016175117,0.0004878992,0.8654434,0.00019628736,0.00026819835,0.00015414634,0.0014551567,0.103719175,0.012100481],"genre_scores_gemma":[0.12575205,0.00019351562,0.83472496,0.00028188745,0.00009576202,0.00033674476,0.0050645065,0.015139713,0.018410863],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99966323,0.000046570385,0.000017799443,0.00009035751,0.00012629454,0.000055724908],"domain_scores_gemma":[0.9996551,0.00015221839,0.000018469866,0.000067438305,0.00008475372,0.000021985152],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00032031906,0.0013657251,0.0007063716,0.0006588208,0.0004909421,0.0009459286,0.0018290291,0.0009651304,0.035540875],"category_scores_gemma":[0.0014029104,0.000787979,0.0010512713,0.0007229983,0.00039209204,0.0013141657,0.0012620414,0.0014561136,0.010189919],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006257081,0.000200216,0.0006878911,0.00038479757,0.00012226032,0.00023595325,0.000114668626,0.058943722,0.04910698,0.013269281,0.15332778,0.7229807],"study_design_scores_gemma":[0.00021163841,0.00008546424,0.00026285605,0.000021024262,0.000042384436,0.000105476494,0.000037963247,0.92381054,0.04618504,0.009154135,0.020056983,0.000026554228],"about_ca_topic_score_codex":0.0071111936,"about_ca_topic_score_gemma":0.015781917,"teacher_disagreement_score":0.035540875,"about_ca_system_score_codex":0.00074941077,"about_ca_system_score_gemma":0.00084315514,"threshold_uncertainty_score":0.11889607},"labels":[],"label_agreement":null},{"id":"W4389518385","doi":"10.18653/v1/2023.arabicnlp-1.1","title":"Violet: A Vision-Language Model for Arabic Image Captioning with Gemini Decoder","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Closed captioning; Computer science; Fluency; Natural language processing; Artificial intelligence; Arabic; Language model; Encoder; Image (mathematics); Generative model; Speech recognition; Generative grammar; Linguistics","score_opus":0.012239302839231678,"score_gpt":0.3186121580308731,"score_spread":0.3063728551916414,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389518385","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031623326,0.0021271626,0.87364733,0.001390772,0.00068574754,0.00045305703,0.005398099,0.06929446,0.015379968],"genre_scores_gemma":[0.4019569,0.0009891656,0.5398353,0.0016109301,0.00032557794,0.0007358083,0.016568638,0.002797885,0.035179853],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996062,0.000083255785,0.000023728373,0.0001499582,0.000085080464,0.00005183761],"domain_scores_gemma":[0.9993098,0.00026266853,0.000040923125,0.00010325257,0.00022781681,0.000055535307],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010680824,0.001670402,0.000817761,0.0009789374,0.000523939,0.0013156609,0.0026642263,0.0014441587,0.007935103],"category_scores_gemma":[0.0027107939,0.0006022694,0.0010179674,0.0006250249,0.0006331254,0.0020246385,0.0013986609,0.00252678,0.0067382753],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00080433005,0.00040586427,0.0019447376,0.00053795514,0.0001940532,0.00038688743,0.00035508643,0.2598088,0.02474299,0.022067314,0.09563872,0.59311324],"study_design_scores_gemma":[0.000030655065,0.00005683928,0.00015827705,0.000018477032,0.000021795779,0.000087314496,0.000027554443,0.97767663,0.00841792,0.0058910344,0.007589601,0.000023993678],"about_ca_topic_score_codex":0.011490939,"about_ca_topic_score_gemma":0.023071712,"teacher_disagreement_score":0.011490939,"about_ca_system_score_codex":0.0016393078,"about_ca_system_score_gemma":0.0013990832,"threshold_uncertainty_score":0.026545584},"labels":[],"label_agreement":null},{"id":"W4389518749","doi":"10.18653/v1/2023.emnlp-main.297","title":"EDIS: Entity-Driven Image Search over Multimodal Web Content","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Waterloo","funders":"","keywords":"Computer science; Information retrieval; Set (abstract data type); Image retrieval; Search engine; Image (mathematics); Ranking (information retrieval); Artificial intelligence","score_opus":0.03877009136996899,"score_gpt":0.3183061662962966,"score_spread":0.2795360749263276,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389518749","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.30972695,0.0095802685,0.5113749,0.0019311943,0.0006121671,0.0029040608,0.09179199,0.053419176,0.018659303],"genre_scores_gemma":[0.43199697,0.0012667064,0.40739462,0.0006622285,0.00029134454,0.0008942826,0.1454333,0.00086494104,0.011195576],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9989673,0.00021211941,0.00008221622,0.0003242434,0.00030857217,0.00010559147],"domain_scores_gemma":[0.99863976,0.0004565641,0.00010567319,0.00045658392,0.00024334787,0.00009802211],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012827672,0.0014896287,0.0012973542,0.0040724734,0.0006010277,0.0014943598,0.0017502335,0.0012797017,0.004747941],"category_scores_gemma":[0.0044765878,0.00028449218,0.0012790827,0.0025971024,0.00042091994,0.003356347,0.0023067289,0.0010580653,0.002870109],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022992296,0.0015414365,0.013931056,0.0026208868,0.00091245346,0.0009809429,0.0004487252,0.05515196,0.054372083,0.013373822,0.17603564,0.6783318],"study_design_scores_gemma":[0.00023128193,0.00063790573,0.010796516,0.000080144986,0.00018589184,0.0012456626,0.000468926,0.8952203,0.036530346,0.012178785,0.042303372,0.00012086571],"about_ca_topic_score_codex":0.009603654,"about_ca_topic_score_gemma":0.017601833,"teacher_disagreement_score":0.009603654,"about_ca_system_score_codex":0.0008922997,"about_ca_system_score_gemma":0.0008608047,"threshold_uncertainty_score":0.01909548},"labels":[],"label_agreement":null},{"id":"W4389519026","doi":"10.18653/v1/2023.emnlp-main.496","title":"GD-COMET: A Geo-Diverse Commonsense Inference Model","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Comet; Commonsense knowledge; Inference; Task (project management); Computer science; Commonsense reasoning; Range (aeronautics); Natural language processing; Artificial intelligence; Data science; Knowledge-based systems; Astronomy; Physics; Engineering; Systems engineering","score_opus":0.05572125529200657,"score_gpt":0.3378330340009139,"score_spread":0.28211177870890736,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389519026","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026461817,0.00047927094,0.9351443,0.003325683,0.00024531534,0.0004424043,0.00492589,0.005727562,0.0232478],"genre_scores_gemma":[0.5462463,0.00041217983,0.43049902,0.0026327197,0.00016376254,0.0007422224,0.010331296,0.000612802,0.0083596995],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99878985,0.00039208718,0.000057961122,0.0004048309,0.0002900794,0.00006522894],"domain_scores_gemma":[0.99608505,0.0026148383,0.0001481608,0.00054111285,0.00040314364,0.00020769736],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001842846,0.00070523855,0.00053116505,0.0013893606,0.0011576915,0.002062504,0.0026435256,0.0017077504,0.007500951],"category_scores_gemma":[0.014325426,0.0003939511,0.0014682589,0.0009487375,0.0015204939,0.004074001,0.0042183693,0.0028788545,0.0014373652],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00067573064,0.00059994217,0.012717472,0.0008787788,0.0005227985,0.0019671337,0.0030832482,0.23794313,0.008088679,0.24282576,0.07809571,0.41260177],"study_design_scores_gemma":[0.000086338,0.00006284471,0.0012380813,0.00009963938,0.00007210976,0.00040390165,0.0003330638,0.7533761,0.0024738812,0.2160774,0.02572697,0.00004971312],"about_ca_topic_score_codex":0.008489027,"about_ca_topic_score_gemma":0.017920256,"teacher_disagreement_score":0.008489027,"about_ca_system_score_codex":0.0012179149,"about_ca_system_score_gemma":0.0016583975,"threshold_uncertainty_score":0.025093138},"labels":[],"label_agreement":null},{"id":"W4389520066","doi":"10.18653/v1/2023.findings-emnlp.46","title":"Video-Text Retrieval by Supervised Sparse Multi-Grained Learning","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Similarity (geometry); Sparse approximation; Artificial intelligence; Frame (networking); Representation (politics); Space (punctuation); Pattern recognition (psychology); Information retrieval; Image (mathematics)","score_opus":0.025579838549131522,"score_gpt":0.2862182801294178,"score_spread":0.26063844158028626,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389520066","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017193308,0.000555181,0.9791915,0.00020035273,0.00004495497,0.00009912024,0.00023815202,0.0016859531,0.00079157244],"genre_scores_gemma":[0.49993974,0.00071648264,0.4868708,0.0008496429,0.00045400494,0.0004829013,0.003331639,0.00035164432,0.0070030997],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991755,0.00021571958,0.00005364352,0.0002516352,0.00021697218,0.00008647768],"domain_scores_gemma":[0.9985238,0.0006771423,0.00021030584,0.00025508186,0.00026296554,0.00007069648],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010610522,0.0010840626,0.0017868349,0.0014063108,0.00046975096,0.00089135946,0.002264222,0.0014063537,0.0024884618],"category_scores_gemma":[0.0046152426,0.00038738266,0.00086203794,0.0019084668,0.0007795057,0.0027243905,0.0015311729,0.0014707429,0.0014699013],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005506715,0.00043584887,0.0011657765,0.00041145663,0.00015424269,0.00015664118,0.00018804274,0.30206722,0.024935465,0.009533147,0.013816354,0.64658517],"study_design_scores_gemma":[0.00002900742,0.00008411284,0.00014868584,0.000005486473,0.000012814215,0.00004448035,0.000019510742,0.99045485,0.0022345886,0.006116358,0.00084030104,0.000009898781],"about_ca_topic_score_codex":0.0058615333,"about_ca_topic_score_gemma":0.008275996,"teacher_disagreement_score":0.0058615333,"about_ca_system_score_codex":0.0007413844,"about_ca_system_score_gemma":0.0009110959,"threshold_uncertainty_score":0.011654794},"labels":[],"label_agreement":null},{"id":"W4389520202","doi":"10.18653/v1/2023.findings-emnlp.60","title":"InvGC: Robust Cross-Modal Retrieval by Inverse Graph Convolution","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Pooling; Modal; Graph; Representation (politics); Convolution (computer science); Theoretical computer science; Algorithm; Artificial intelligence; Artificial neural network","score_opus":0.022764599988271933,"score_gpt":0.2924877190307626,"score_spread":0.2697231190424907,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389520202","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029873217,0.0015816427,0.9575103,0.00027424216,0.000115874085,0.0001409583,0.00050890166,0.007533333,0.002461674],"genre_scores_gemma":[0.3601391,0.0012876098,0.6241284,0.0009543209,0.00020265928,0.00031734924,0.0048253234,0.0012642433,0.006880966],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99842924,0.0002826912,0.00010013221,0.00047827364,0.0005237474,0.00018595118],"domain_scores_gemma":[0.99856913,0.0003646119,0.00013505547,0.0005034596,0.00034819444,0.000079515885],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020755748,0.0017350115,0.002043394,0.0029994154,0.00081184274,0.0022166069,0.0025094121,0.0018605404,0.002933484],"category_scores_gemma":[0.005530988,0.0005114602,0.0018004939,0.0033089344,0.0010794444,0.0036346102,0.0031113164,0.0015695564,0.0021919268],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055495236,0.0002647546,0.0020152833,0.0004450664,0.00038102907,0.00028863677,0.00024893793,0.10621814,0.04555796,0.011113595,0.016196204,0.8167155],"study_design_scores_gemma":[0.00004976299,0.00019962285,0.001329773,0.000024607934,0.000097325705,0.000490879,0.000117444244,0.95468587,0.019992283,0.017350549,0.005592062,0.0000697857],"about_ca_topic_score_codex":0.009711521,"about_ca_topic_score_gemma":0.010498821,"teacher_disagreement_score":0.009711521,"about_ca_system_score_codex":0.0010045244,"about_ca_system_score_gemma":0.0016049678,"threshold_uncertainty_score":0.019309998},"labels":[],"label_agreement":null},{"id":"W4389520321","doi":"10.18653/v1/2023.emnlp-main.735","title":"Visually-Situated Natural Language Understanding with Contrastive Reading Model and Frozen Large Language Models","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Computer science; Situated; Natural park; Reading (process); Natural language; Natural language processing; Linguistics; Artificial intelligence; Philosophy; Ecology","score_opus":0.02192735941460207,"score_gpt":0.2936765613624693,"score_spread":0.2717492019478672,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389520321","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13394436,0.00027651107,0.8550782,0.0006204636,0.00006723295,0.00009821222,0.0003073454,0.002145404,0.0074622715],"genre_scores_gemma":[0.884248,0.00009543235,0.11256396,0.00006308632,0.000022871982,0.00009524432,0.00045734108,0.00025869656,0.0021952318],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996164,0.0001566465,0.000018259412,0.00013972318,0.000034079254,0.000034951157],"domain_scores_gemma":[0.9976909,0.0014850267,0.00017106273,0.000309208,0.0002017456,0.00014212917],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009776852,0.0006751688,0.00077409897,0.0006867436,0.0004844065,0.0024597333,0.0016644123,0.0010704456,0.004632989],"category_scores_gemma":[0.00592598,0.0006246798,0.0013225866,0.00036578646,0.0008744406,0.006265955,0.0019301095,0.002006625,0.0010100252],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011641127,0.00072493986,0.0057564615,0.0003142593,0.00028020807,0.00093396014,0.003374024,0.5668233,0.01821568,0.14034441,0.0059919525,0.25607666],"study_design_scores_gemma":[0.000021820437,0.000030154219,0.00023827994,0.0000065967365,0.000014391528,0.000022370874,0.00011304127,0.9581128,0.0008512882,0.040345985,0.00023125872,0.000012050953],"about_ca_topic_score_codex":0.012297369,"about_ca_topic_score_gemma":0.013448566,"teacher_disagreement_score":0.012297369,"about_ca_system_score_codex":0.0012258849,"about_ca_system_score_gemma":0.0009118636,"threshold_uncertainty_score":0.024451554},"labels":[],"label_agreement":null},{"id":"W4389520394","doi":"10.18653/v1/2023.emnlp-main.652","title":"Balance Act: Mitigating Hubness in Cross-Modal Retrieval with Query and Gallery Banks","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Softmax function; Inference; Boosting (machine learning); Information retrieval; Normalization (sociology); Similarity (geometry); Dual (grammatical number); Data mining; Artificial intelligence; Image (mathematics); Deep learning","score_opus":0.011786407206670723,"score_gpt":0.2904348914203238,"score_spread":0.27864848421365307,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389520394","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06693676,0.0013215704,0.9204487,0.00042095955,0.00015923806,0.00030780825,0.0004720539,0.006741966,0.0031908993],"genre_scores_gemma":[0.63125265,0.000741894,0.34671903,0.0012913973,0.00037854785,0.00044560898,0.0025652882,0.001013376,0.015592236],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997343,0.0005778743,0.00019357924,0.0007727173,0.00080127554,0.00031162935],"domain_scores_gemma":[0.9958254,0.0014829682,0.000348447,0.0012007197,0.00096313964,0.00017942554],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004581484,0.0018216573,0.0022098792,0.002642765,0.001372023,0.0022638384,0.0027055936,0.0019409613,0.0050987853],"category_scores_gemma":[0.012789772,0.00069743965,0.0011754318,0.002212683,0.0016953867,0.005071864,0.004056308,0.0021168997,0.0033767154],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00241947,0.00073566544,0.0059779813,0.0005749583,0.00037873036,0.0003792407,0.00096293306,0.054426245,0.08122078,0.009516234,0.018585136,0.82482266],"study_design_scores_gemma":[0.00014464337,0.0005104007,0.005940326,0.000059924514,0.00029647324,0.0005026221,0.00053785683,0.89740074,0.066009626,0.018528288,0.009943168,0.000125936],"about_ca_topic_score_codex":0.0056627053,"about_ca_topic_score_gemma":0.00845239,"teacher_disagreement_score":0.0056627053,"about_ca_system_score_codex":0.0010483464,"about_ca_system_score_gemma":0.0017792603,"threshold_uncertainty_score":0.024229467},"labels":[],"label_agreement":null},{"id":"W4389521029","doi":"10.18653/v1/2023.conll-1.7","title":"ArchBERT: Bi-Modal Understanding of Neural Architectures and Natural Languages","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada)","funders":"","keywords":"Computer science; Automatic summarization; Natural language; Architecture; Artificial intelligence; Closed captioning; Artificial neural network; Modal; Set (abstract data type); Modalities; Natural language processing; Modality (human–computer interaction); Question answering; Natural language understanding; Inference; Programming language; Image (mathematics)","score_opus":0.023390957982274044,"score_gpt":0.30562016566167766,"score_spread":0.2822292076794036,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389521029","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00819425,0.0002773258,0.9829707,0.0005550668,0.00004133634,0.000055557808,0.0015150063,0.0036191654,0.002771685],"genre_scores_gemma":[0.37436354,0.0006272447,0.6006102,0.00084978045,0.0001152066,0.00067486364,0.007817414,0.0008676316,0.014074186],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99957734,0.00012714593,0.000018686344,0.00016308507,0.00006992809,0.000043814875],"domain_scores_gemma":[0.99908006,0.00042905132,0.000082525985,0.00023890575,0.00011777034,0.000051669107],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007620331,0.0010346255,0.00043680985,0.0008300964,0.00050647685,0.001596207,0.0022893983,0.0016391245,0.0068871626],"category_scores_gemma":[0.0035923838,0.0006960833,0.001467226,0.00065824745,0.00071466283,0.0038745587,0.0020812163,0.0036585112,0.0016255907],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002903584,0.00020632501,0.00262991,0.00038981883,0.00024260224,0.00026905877,0.0006756293,0.50980395,0.010334822,0.16363828,0.027112722,0.28440654],"study_design_scores_gemma":[0.000008068447,0.0000186812,0.00024138812,0.000017662851,0.000011879284,0.00003965782,0.00002709152,0.92132145,0.0012152432,0.07265981,0.0044273552,0.000011680723],"about_ca_topic_score_codex":0.009214528,"about_ca_topic_score_gemma":0.021464236,"teacher_disagreement_score":0.009214528,"about_ca_system_score_codex":0.0012431827,"about_ca_system_score_gemma":0.00094608765,"threshold_uncertainty_score":0.023039877},"labels":[],"label_agreement":null},{"id":"W4389523744","doi":"10.18653/v1/2023.emnlp-main.629","title":"APoLLo : Unified Adapter and Prompt Learning for Vision Language Models","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Army Research Office","keywords":"Computer science; Adapter (computing); Generalization; Artificial intelligence; Overfitting; Encoder; Modalities; Machine learning; Artificial neural network","score_opus":0.022641774636371664,"score_gpt":0.30940200390185413,"score_spread":0.28676022926548245,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389523744","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034390673,0.0008592684,0.8860439,0.0005104641,0.00027248065,0.00028124757,0.0011899689,0.072117984,0.004334013],"genre_scores_gemma":[0.48442823,0.00044099623,0.4925225,0.0015635724,0.00019157892,0.0008878429,0.005932606,0.0020542673,0.011978318],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991972,0.00016068124,0.00003182212,0.00036481707,0.00012929484,0.00011628643],"domain_scores_gemma":[0.9987212,0.00045044266,0.00006612005,0.00041860773,0.00021263723,0.00013105973],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020048944,0.0022266454,0.0011466421,0.00080324925,0.0005638172,0.0014092556,0.003919529,0.0019825632,0.008461995],"category_scores_gemma":[0.005308883,0.00067632773,0.0012946861,0.0006309051,0.0007573781,0.0038977265,0.0042339573,0.0041904696,0.0033405214],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007691487,0.0006366517,0.0021356058,0.00032843952,0.00021627483,0.00029415407,0.0002457844,0.060179096,0.02684979,0.006804783,0.039995503,0.8615447],"study_design_scores_gemma":[0.00007957227,0.00024393461,0.0005456594,0.000029077542,0.00004434574,0.00012387484,0.000084701096,0.96474487,0.016395796,0.011683547,0.005987389,0.000037310318],"about_ca_topic_score_codex":0.005999408,"about_ca_topic_score_gemma":0.010674329,"teacher_disagreement_score":0.008461995,"about_ca_system_score_codex":0.0011168694,"about_ca_system_score_gemma":0.0017581701,"threshold_uncertainty_score":0.028308153},"labels":[],"label_agreement":null},{"id":"W4389523846","doi":"10.18653/v1/2023.emnlp-main.906","title":"UniChart: A Universal Vision-language Pretrained Model for Chart Comprehension and Reasoning","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":49,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Chart; Computer science; Automatic summarization; Artificial intelligence; Natural language processing; Generalizability theory; Question answering; Machine learning","score_opus":0.01654483011686048,"score_gpt":0.3042212782774124,"score_spread":0.28767644816055193,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389523846","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05154606,0.0011321271,0.89106643,0.00081058725,0.0003677947,0.0005197893,0.0048569697,0.04254909,0.0071511017],"genre_scores_gemma":[0.43580428,0.00085805595,0.5234191,0.0011964493,0.00018395063,0.001311805,0.01860002,0.0011217477,0.017504632],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99957997,0.00006600577,0.00002299961,0.00021529383,0.0000653189,0.000050496004],"domain_scores_gemma":[0.9988686,0.0005669909,0.000070898226,0.0001467298,0.00027471592,0.00007215633],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00079419214,0.0018599373,0.00078461884,0.0009822643,0.00040887465,0.0013496953,0.0028301845,0.0015476474,0.00863396],"category_scores_gemma":[0.0038902797,0.00064093183,0.001533309,0.0006934759,0.0006387256,0.0023849686,0.0011838018,0.0030414215,0.0029674752],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005814395,0.00048910215,0.0026035097,0.00050131965,0.00016267026,0.00036276513,0.00034402133,0.26513976,0.025108669,0.009099512,0.041467074,0.6541402],"study_design_scores_gemma":[0.00002416486,0.00007434185,0.000321013,0.000021937276,0.000025645366,0.000033748092,0.000025954052,0.9864319,0.005332223,0.0044466713,0.0032469823,0.00001534757],"about_ca_topic_score_codex":0.014830944,"about_ca_topic_score_gemma":0.023024455,"teacher_disagreement_score":0.014830944,"about_ca_system_score_codex":0.0015407673,"about_ca_system_score_gemma":0.0021382917,"threshold_uncertainty_score":0.02948922},"labels":[],"label_agreement":null},{"id":"W4390481874","doi":"10.1109/ssci52147.2023.10371820","title":"Image Caption Generation Based on Image-Text Matching Schema in Deep Reinforcement Learning","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Closed captioning; Reinforcement learning; Computer science; Artificial intelligence; Matching (statistics); Schema (genetic algorithms); Image (mathematics); Deep learning; Machine learning; Computer vision","score_opus":0.018295307871210587,"score_gpt":0.28744745214003,"score_spread":0.2691521442688194,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390481874","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029559828,0.00056533224,0.9565857,0.00037510708,0.00021538163,0.0002872836,0.0002623859,0.007587954,0.0045611216],"genre_scores_gemma":[0.5359486,0.0003553818,0.453346,0.0004393237,0.00010988048,0.00035504546,0.0010217866,0.000641662,0.007782295],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99947006,0.00013187116,0.000026846468,0.00019996289,0.00011245779,0.000058876398],"domain_scores_gemma":[0.998931,0.00044489407,0.00010479493,0.00018309217,0.0002629193,0.000073307136],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011615681,0.0011004908,0.0009454114,0.00073211355,0.0004187929,0.0010867671,0.0016003416,0.0016058963,0.00488582],"category_scores_gemma":[0.0041070906,0.00043179982,0.00061099284,0.00068640657,0.0007475815,0.0016550984,0.0010278627,0.0016365662,0.0013835318],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003641086,0.00032513397,0.0009767694,0.0002883361,0.000071373535,0.00034977938,0.0002195118,0.52999055,0.02690965,0.010970386,0.015563287,0.41397113],"study_design_scores_gemma":[0.000014366683,0.000023040318,0.000070054186,0.0000061547025,0.0000059537933,0.000026131667,0.000008138089,0.9905963,0.00586892,0.002525224,0.0008491931,0.0000064675846],"about_ca_topic_score_codex":0.0038593286,"about_ca_topic_score_gemma":0.0036150925,"teacher_disagreement_score":0.00488582,"about_ca_system_score_codex":0.0011295833,"about_ca_system_score_gemma":0.0008244343,"threshold_uncertainty_score":0.016344726},"labels":[],"label_agreement":null},{"id":"W4390708079","doi":"10.1007/978-3-031-23161-2_501","title":"Automated Image Captioning for the Visually Impaired","year":2024,"lang":"en","type":"book-chapter","venue":"Encyclopedia of Computer Graphics and Games","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"CNIB Foundation; Ontario Tech University","funders":"","keywords":"Closed captioning; Visually impaired; Computer vision; Computer science; Image (mathematics); Artificial intelligence; Computer graphics (images); Human–computer interaction","score_opus":0.009772799513870566,"score_gpt":0.2649087258365698,"score_spread":0.25513592632269927,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390708079","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.082237706,0.014074802,0.719971,0.0026084448,0.002571069,0.00077964383,0.0051863743,0.032563698,0.14000732],"genre_scores_gemma":[0.2659682,0.014205531,0.5506765,0.00072165555,0.00086651,0.00048724664,0.009417747,0.003081565,0.15457508],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9998307,0.000040534498,0.000010807029,0.000031258085,0.00006689672,0.00001974151],"domain_scores_gemma":[0.9995629,0.0001561504,0.000020161116,0.00006416497,0.0001731217,0.000023477765],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00029655732,0.00086301635,0.00038819248,0.0010618935,0.00032772895,0.0014539064,0.0009157555,0.00093197863,0.032017443],"category_scores_gemma":[0.0017542244,0.00023883324,0.00037065434,0.0005710887,0.00035365735,0.0010316535,0.0008727845,0.0006008656,0.009883289],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023552112,0.00007462512,0.00021595697,0.00038162555,0.000014062582,0.00037553097,0.00016430848,0.0033495808,0.024927892,0.0022519678,0.10313777,0.8648712],"study_design_scores_gemma":[0.000115840565,0.0007771685,0.008808548,0.0009778519,0.00018122699,0.007421153,0.0013704038,0.26466018,0.15759365,0.026147652,0.5317496,0.00019677314],"about_ca_topic_score_codex":0.0024175593,"about_ca_topic_score_gemma":0.0023087347,"teacher_disagreement_score":0.032017443,"about_ca_system_score_codex":0.00026761316,"about_ca_system_score_gemma":0.00040489968,"threshold_uncertainty_score":0.10710901},"labels":[],"label_agreement":null},{"id":"W4390873290","doi":"10.1109/iccv51070.2023.01282","title":"Tem-adapter: Adapting Image-Text Pretraining for Video Question Answer","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"HORIZON EUROPE Health; Tsinghua Shenzhen International Graduate School; National Natural Science Foundation of China","keywords":"Computer science; Adapter (computing); Leverage (statistics); Transformer; Artificial intelligence; Natural language processing; Event (particle physics); Semantics (computer science); Feature learning; Task (project management); Speech recognition; Programming language","score_opus":0.025625017955884403,"score_gpt":0.319102130015953,"score_spread":0.2934771120600686,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390873290","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05245527,0.0005138357,0.8837852,0.0003346389,0.00024750954,0.0005493059,0.0015078619,0.055820722,0.004785618],"genre_scores_gemma":[0.2920625,0.00030130675,0.6847343,0.00085783936,0.00013919082,0.0008496786,0.008260912,0.0013237857,0.011470536],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.999474,0.000109860375,0.000027737928,0.00025810825,0.00007515154,0.00005515756],"domain_scores_gemma":[0.99873143,0.00067538134,0.00006996153,0.00020242823,0.00025600748,0.000064803135],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010342179,0.0018107273,0.0006592815,0.0009175739,0.00031169446,0.0006425927,0.0031255616,0.0020939012,0.009358732],"category_scores_gemma":[0.004443407,0.000543414,0.00096853444,0.00072114414,0.00045145646,0.0022298638,0.0014313584,0.0020474603,0.0034821185],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00054859964,0.0007270277,0.0018825841,0.00034516313,0.00013426004,0.00023132838,0.00029076546,0.06280142,0.04353371,0.0023552252,0.026813157,0.8603369],"study_design_scores_gemma":[0.000064861844,0.00021310912,0.0008146496,0.000019528474,0.000040260504,0.000089234985,0.00012456806,0.9577263,0.032214005,0.0026059404,0.006062259,0.000025329824],"about_ca_topic_score_codex":0.008829944,"about_ca_topic_score_gemma":0.012144315,"teacher_disagreement_score":0.009358732,"about_ca_system_score_codex":0.00089199113,"about_ca_system_score_gemma":0.00088588806,"threshold_uncertainty_score":0.031308055},"labels":[],"label_agreement":null},{"id":"W4390874807","doi":"10.1109/acii59096.2023.10388198","title":"Contextual Emotion Estimation from Image Captions","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Estimation; Image (mathematics); Artificial intelligence; Natural language processing; Computer vision; Engineering","score_opus":0.017812830750340628,"score_gpt":0.2950309506352097,"score_spread":0.27721811988486905,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390874807","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1644908,0.004408272,0.77200586,0.0018932133,0.001121503,0.00090798654,0.013337187,0.026057951,0.015777275],"genre_scores_gemma":[0.60211456,0.0015292965,0.3600555,0.0010893058,0.0006546755,0.00061871094,0.025407815,0.0014129692,0.007117114],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991542,0.0002713916,0.000040048857,0.00032606957,0.00012359302,0.00008471773],"domain_scores_gemma":[0.9979882,0.00077739434,0.0002213846,0.00042613136,0.00051899883,0.00006793819],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011146113,0.0021081534,0.0006691151,0.0011387445,0.0003750408,0.0013537607,0.0010738514,0.0012175703,0.0039275107],"category_scores_gemma":[0.0060199485,0.00036664496,0.0013623474,0.00060403196,0.0006142796,0.0022284118,0.000972561,0.0017381062,0.0027091803],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020570382,0.00038506454,0.009658197,0.0016965559,0.0004477455,0.0010423867,0.0010906962,0.089733675,0.15953763,0.0096975975,0.09033896,0.6343144],"study_design_scores_gemma":[0.00005958359,0.00037000584,0.0091023855,0.00020909557,0.0001745783,0.0006497317,0.0005653732,0.87401253,0.06643232,0.0134379845,0.03489697,0.000089429195],"about_ca_topic_score_codex":0.0019509955,"about_ca_topic_score_gemma":0.0028519565,"teacher_disagreement_score":0.0039275107,"about_ca_system_score_codex":0.00086510845,"about_ca_system_score_gemma":0.00036548427,"threshold_uncertainty_score":0.013138771},"labels":[],"label_agreement":null},{"id":"W4391173394","doi":"10.1117/1.jei.33.1.013028","title":"TEG: image theme recognition using text-embedding-guided few-shot adaptation","year":2024,"lang":"en","type":"article","venue":"Journal of Electronic Imaging","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Computer science; Artificial intelligence; Computer vision; Theme (computing); Image processing; Embedding; Adaptation (eye); Pattern recognition (psychology); Image (mathematics); Optics; Physics","score_opus":0.036764065489061935,"score_gpt":0.3436573871034038,"score_spread":0.3068933216143419,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391173394","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10257141,0.0031589246,0.8417926,0.0005540927,0.0010759017,0.0006071154,0.0040573333,0.038252477,0.007930146],"genre_scores_gemma":[0.39701468,0.0012159833,0.5469445,0.00095759257,0.00043429824,0.00052307366,0.018022342,0.0017007056,0.033186838],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995987,0.000040816223,0.000015417658,0.00021254942,0.000079223,0.000053402648],"domain_scores_gemma":[0.9995615,0.00006465856,0.000031567204,0.00017722743,0.00011746303,0.000047577178],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00047490362,0.0013724811,0.0010237523,0.0011228968,0.00041461465,0.0007087296,0.0021865459,0.0010400827,0.0049733184],"category_scores_gemma":[0.0014033768,0.00027856,0.0010780103,0.000870447,0.00045973822,0.0018532339,0.0012634991,0.0015570727,0.0047267876],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006313526,0.0003504425,0.0012560821,0.00039858025,0.00018387292,0.00036802227,0.00018532388,0.016457807,0.123573214,0.0017451491,0.04168424,0.81316584],"study_design_scores_gemma":[0.00008561648,0.0005830672,0.003924246,0.000051143244,0.00014255002,0.0011051983,0.00023841423,0.84562886,0.11647221,0.007067248,0.02461123,0.00009016904],"about_ca_topic_score_codex":0.0034666243,"about_ca_topic_score_gemma":0.0065653566,"teacher_disagreement_score":0.0049733184,"about_ca_system_score_codex":0.00046332445,"about_ca_system_score_gemma":0.00058977824,"threshold_uncertainty_score":0.016637444},"labels":[],"label_agreement":null},{"id":"W4391335180","doi":"10.1145/3610977.3634999","title":"Generative Expressive Robot Behaviors using Large Language Models","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Robot; Human–computer interaction; Leverage (statistics); Motion (physics); Generative model; Generative grammar; Natural language; Context (archaeology); Artificial intelligence; Social robot; Mobile robot; Robot control","score_opus":0.03655862504537216,"score_gpt":0.3494611117712703,"score_spread":0.31290248672589815,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391335180","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020709664,0.00017885001,0.97118515,0.0002407902,0.000038220856,0.00009985699,0.00033322032,0.0038093133,0.003405003],"genre_scores_gemma":[0.6010007,0.00025344922,0.38598412,0.00028344378,0.00003992196,0.0006163047,0.0015331685,0.0012267089,0.009062151],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995915,0.0001597804,0.000017368835,0.00012722595,0.00007505273,0.000028993163],"domain_scores_gemma":[0.9990773,0.000656114,0.00005316706,0.00011080184,0.000063703555,0.000038916998],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00056548155,0.00096226484,0.00046407478,0.0003991587,0.0003813935,0.00091478846,0.0010988169,0.00088564755,0.005343778],"category_scores_gemma":[0.002722764,0.0006405975,0.0012297168,0.00025242078,0.0008789808,0.0009891952,0.0013317348,0.0011750332,0.0014563958],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001302914,0.000113612914,0.0014160266,0.00021914633,0.0000772664,0.0003712639,0.00071978004,0.86582816,0.014282533,0.023136996,0.004154562,0.08955048],"study_design_scores_gemma":[0.000013305719,0.00002137766,0.00011392197,0.00001115243,0.0000068164723,0.00003563071,0.000036712598,0.9858262,0.0014615656,0.010766338,0.0016974908,0.000009502633],"about_ca_topic_score_codex":0.0035825623,"about_ca_topic_score_gemma":0.008375501,"teacher_disagreement_score":0.005343778,"about_ca_system_score_codex":0.0007256393,"about_ca_system_score_gemma":0.00057994324,"threshold_uncertainty_score":0.017876685},"labels":[],"label_agreement":null},{"id":"W4391457730","doi":"10.5206/fpq/2022.3/4.14292","title":"Bias Dilemma","year":2022,"lang":"en","type":"article","venue":"Feminist Philosophy Quarterly","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Dilemma; Computer science; Epistemology; Philosophy","score_opus":0.0307340521760262,"score_gpt":0.2603152871605439,"score_spread":0.2295812349845177,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391457730","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014762535,0.0051494646,0.16470513,0.53832954,0.0040114643,0.00023881174,0.00027661465,0.00035375537,0.2721727],"genre_scores_gemma":[0.63615674,0.004060212,0.08926385,0.21658528,0.006883809,0.0010321554,0.00022116394,0.0006622402,0.045134578],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9585999,0.023253536,0.0017124772,0.0045873914,0.010582214,0.0012644705],"domain_scores_gemma":[0.93132013,0.049129903,0.0023915197,0.006786115,0.008924715,0.001447661],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.048278373,0.00081186515,0.000990147,0.0017543841,0.004733713,0.0069723874,0.0026557434,0.009341687,0.009896498],"category_scores_gemma":[0.1121767,0.00047907385,0.0011149262,0.0010821745,0.032595653,0.015256149,0.006563273,0.014013796,0.0031380472],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000022198801,0.0000135904465,0.00032132652,0.00006705254,0.000017960394,0.000055504806,0.0010757436,0.00029683768,0.00016992523,0.975661,0.011059703,0.011239226],"study_design_scores_gemma":[0.000022916525,0.000011187619,0.000083524006,0.000090318645,0.0000089577115,0.000096705655,0.00028232238,0.0005215429,0.00023573784,0.9442955,0.054337326,0.000013932767],"about_ca_topic_score_codex":0.0015932013,"about_ca_topic_score_gemma":0.0010383481,"teacher_disagreement_score":0.048278373,"about_ca_system_score_codex":0.004646259,"about_ca_system_score_gemma":0.0067525776,"threshold_uncertainty_score":0.25532353},"labels":[],"label_agreement":null},{"id":"W4391516313","doi":"10.1007/978-981-99-8850-1_25","title":"PointerNet with Local and Global Contexts for Natural Language Moment Localization","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Moment (physics); Natural (archaeology); Natural language; Natural language processing; Artificial intelligence; Physics; Geography; Archaeology","score_opus":0.005756129519541871,"score_gpt":0.26191497201004543,"score_spread":0.2561588424905036,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391516313","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006792826,0.00047169748,0.9725929,0.00013303861,0.00017422142,0.00004646217,0.0005276281,0.014769542,0.0044916724],"genre_scores_gemma":[0.24079995,0.0006454692,0.7390246,0.00026267118,0.0002283191,0.00021993971,0.0028503889,0.001953322,0.014015338],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99928015,0.00010443162,0.00005742559,0.00029383413,0.00017753341,0.00008673669],"domain_scores_gemma":[0.9991702,0.0002595622,0.000045453107,0.00030616412,0.00016462094,0.000053966345],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008209895,0.0011776985,0.0011916902,0.001999379,0.000993619,0.0019908538,0.0019624357,0.0017296349,0.014371895],"category_scores_gemma":[0.002895503,0.0007018011,0.0009047152,0.0017728363,0.0011746096,0.0066234134,0.0034915328,0.001727314,0.005252477],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004978059,0.00013358037,0.0009780814,0.00041230122,0.000072248506,0.000270626,0.00029792608,0.034050774,0.017931791,0.2005751,0.024145678,0.72063416],"study_design_scores_gemma":[0.000052108884,0.00015199684,0.00043712746,0.0001273474,0.000082280574,0.00033521862,0.0001587042,0.6151909,0.019475488,0.32553995,0.038383793,0.00006508122],"about_ca_topic_score_codex":0.0054827626,"about_ca_topic_score_gemma":0.010121172,"teacher_disagreement_score":0.014371895,"about_ca_system_score_codex":0.0008049173,"about_ca_system_score_gemma":0.0013179459,"threshold_uncertainty_score":0.048078775},"labels":[],"label_agreement":null},{"id":"W4391654261","doi":"10.2196/32690","title":"Vision-Language Model for Generating Textual Descriptions From Clinical Images: Model Development and Validation Study","year":2024,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"China Postdoctoral Science Foundation; National Natural Science Foundation of China","keywords":"Computer science; Natural language processing; Artificial intelligence; Standardization; Deep learning; Metric (unit); Closed captioning; Quality (philosophy); Information retrieval; Machine learning; Image (mathematics)","score_opus":0.15177448273116914,"score_gpt":0.505880174163475,"score_spread":0.35410569143230586,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391654261","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.41162506,0.0033181917,0.56272763,0.001456818,0.0003781934,0.0013173353,0.003026528,0.009001731,0.0071484707],"genre_scores_gemma":[0.81579477,0.0006735513,0.17447633,0.00038047248,0.000061464234,0.00081877597,0.004240381,0.00014517891,0.0034091382],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99893194,0.0003864199,0.00007134265,0.00030409195,0.00020001635,0.00010617015],"domain_scores_gemma":[0.9946102,0.0037074122,0.00021146087,0.00029518377,0.0010774578,0.000098356446],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003706851,0.0012768711,0.00076492474,0.0010847037,0.00035609098,0.0010702111,0.0021222734,0.0013090594,0.0030109459],"category_scores_gemma":[0.0094665475,0.00048636596,0.0012532526,0.0006616924,0.00043461187,0.0011701549,0.00093163777,0.002397399,0.0010395888],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006946594,0.0007312707,0.0057279933,0.000451925,0.00022280386,0.00020446647,0.00017860298,0.7004835,0.0041107605,0.0018835159,0.005030508,0.28028002],"study_design_scores_gemma":[0.000016928709,0.000072618386,0.0003270744,0.000009477271,0.000016954185,0.000019301266,0.0000130903645,0.9980299,0.0010023315,0.00025612622,0.00022963516,0.000006634823],"about_ca_topic_score_codex":0.038189918,"about_ca_topic_score_gemma":0.025212472,"teacher_disagreement_score":0.038189918,"about_ca_system_score_codex":0.0020350087,"about_ca_system_score_gemma":0.0021258034,"threshold_uncertainty_score":0.075935245},"labels":[],"label_agreement":null},{"id":"W4391871827","doi":"10.48550/arxiv.2402.08532","title":"Captions Are Worth a Thousand Words: Enhancing Product Retrieval with Pretrained Image-to-Text Models","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Mitacs","keywords":"Product (mathematics); Computer science; Image (mathematics); Natural language processing; Information retrieval; Artificial intelligence; Mathematics","score_opus":0.03925230774777872,"score_gpt":0.2080312265532537,"score_spread":0.168778918805475,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391871827","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24591027,0.007256508,0.69397974,0.002592399,0.0011389604,0.00078975304,0.0059890314,0.023382148,0.018961117],"genre_scores_gemma":[0.69483185,0.0021523405,0.2654546,0.0021919124,0.0007586343,0.0005374482,0.013934796,0.0011440177,0.01899442],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995553,0.00011504731,0.000026573973,0.0001644104,0.00009484758,0.00004377148],"domain_scores_gemma":[0.99783224,0.0011900206,0.00018804138,0.00033029623,0.00039155287,0.000067818255],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010905697,0.0015126178,0.0007196578,0.0013712752,0.00032629858,0.0011947397,0.0014591181,0.0015483533,0.0035823784],"category_scores_gemma":[0.0060564512,0.0003859278,0.0009972625,0.0011513113,0.00065254536,0.002698661,0.0007743111,0.001599025,0.004100256],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011724696,0.0006828393,0.0035553304,0.00078199373,0.00022317437,0.00057986093,0.00027759618,0.1577702,0.048594553,0.0032040393,0.04722227,0.7359357],"study_design_scores_gemma":[0.000046900288,0.00025588807,0.0012397878,0.000043949512,0.00008493761,0.00025844705,0.00005069965,0.9704206,0.018357176,0.0025747053,0.006631611,0.000035123645],"about_ca_topic_score_codex":0.0067179,"about_ca_topic_score_gemma":0.0073999343,"teacher_disagreement_score":0.0067179,"about_ca_system_score_codex":0.0009405925,"about_ca_system_score_gemma":0.0005216932,"threshold_uncertainty_score":0.01335758},"labels":[],"label_agreement":null},{"id":"W4392384263","doi":"10.1145/3616855.3635757","title":"Text-Video Retrieval via Multi-Modal Hypergraph Networks","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Wilfrid Laurier University","funders":"","keywords":"Computer science; Hypergraph; Matching (statistics); Artificial intelligence; Representation (politics); Visual Word; Pipeline (software); Information retrieval; Pattern recognition (psychology); Natural language processing; Image retrieval; Image (mathematics)","score_opus":0.011400809876122384,"score_gpt":0.2742944262598563,"score_spread":0.2628936163837339,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392384263","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023307512,0.0014818087,0.9700436,0.0005712299,0.000050326915,0.00013388321,0.00042918773,0.0014075089,0.002575065],"genre_scores_gemma":[0.73362315,0.0021560262,0.24626009,0.0011046182,0.00034220243,0.00047461697,0.0028978589,0.0003988078,0.012742549],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99908817,0.00026509512,0.0000426334,0.000323268,0.00019621878,0.00008454328],"domain_scores_gemma":[0.9986739,0.000833431,0.00017699678,0.00011464361,0.00015278082,0.00004822468],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00088779686,0.0010652777,0.0012094035,0.0023940883,0.0007352109,0.0015160851,0.0018457709,0.0021842238,0.002917991],"category_scores_gemma":[0.0040418976,0.00063131057,0.0015839739,0.0024593205,0.00084365346,0.003729333,0.0013244426,0.0013863655,0.0010113881],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046144213,0.0002451526,0.0016373115,0.0004434832,0.00027914788,0.000480933,0.00038800554,0.6412319,0.013174151,0.03394322,0.012371149,0.29534408],"study_design_scores_gemma":[0.00000861937,0.000017329408,0.00015122369,0.0000066685166,0.000017868882,0.00003376027,0.000016649501,0.98890185,0.00074803137,0.009471219,0.0006179456,0.000008919377],"about_ca_topic_score_codex":0.01600359,"about_ca_topic_score_gemma":0.0135813905,"teacher_disagreement_score":0.01600359,"about_ca_system_score_codex":0.0019688678,"about_ca_system_score_gemma":0.0009556595,"threshold_uncertainty_score":0.031820834},"labels":[],"label_agreement":null},{"id":"W4392680666","doi":"10.1145/3613904.3642752","title":"AQuA: Automated Question-Answering in Software Tutorial Videos with Visual Anchors","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Autodesk (Canada)","funders":"","keywords":"Computer science; Documentation; Pipeline (software); Feature (linguistics); Software; Human–computer interaction; Multimedia; Question answering; Information retrieval; World Wide Web; Programming language","score_opus":0.007411309493075668,"score_gpt":0.3091428584012771,"score_spread":0.3017315489082014,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392680666","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1418548,0.0013166623,0.5776062,0.0010122102,0.00036340035,0.0028620667,0.026298922,0.24078207,0.007903648],"genre_scores_gemma":[0.29225048,0.0003216007,0.65291315,0.0003884828,0.00014153839,0.002600757,0.039298527,0.0028165348,0.009268936],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99763787,0.001085311,0.00013104548,0.00062105234,0.00037115763,0.00015355628],"domain_scores_gemma":[0.99395704,0.0039087795,0.00036695704,0.00051912235,0.0009098459,0.00033821372],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028558427,0.0018363711,0.00079169555,0.002925659,0.0005286965,0.0011098825,0.0013067474,0.001579915,0.01561031],"category_scores_gemma":[0.014138894,0.00048436865,0.00081744,0.0009942959,0.00043321916,0.002586601,0.0028378156,0.0011401263,0.007871689],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016844707,0.00072245544,0.0076532615,0.0025321033,0.00016812664,0.0006844664,0.005911078,0.0050752307,0.06877969,0.0028634006,0.1198568,0.7840689],"study_design_scores_gemma":[0.00074929756,0.0020881312,0.03754775,0.00058046944,0.00020572884,0.0015608342,0.006932858,0.627959,0.10421966,0.021087075,0.19671471,0.00035449222],"about_ca_topic_score_codex":0.00304603,"about_ca_topic_score_gemma":0.0051816567,"teacher_disagreement_score":0.01561031,"about_ca_system_score_codex":0.0005873291,"about_ca_system_score_gemma":0.00085262966,"threshold_uncertainty_score":0.052221656},"labels":[],"label_agreement":null},{"id":"W4392864334","doi":"10.1145/3649896","title":"Realizing Efficient On-Device Language-based Image Retrieval","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Multimedia Computing Communications and Applications","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Centre for Social Innovation","funders":"","keywords":"Computer science; Information retrieval; Image retrieval; Latency (audio); Ranking (information retrieval); Modal; Language model; Deep learning; Context (archaeology); Artificial intelligence; Image (mathematics)","score_opus":0.02008231110073879,"score_gpt":0.32669562855509837,"score_spread":0.3066133174543596,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392864334","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07013846,0.0021938754,0.8932771,0.00053396524,0.00021156622,0.00027152637,0.0007735129,0.016708745,0.015891258],"genre_scores_gemma":[0.62890863,0.00093179534,0.34994444,0.0008870829,0.00012611147,0.0002605522,0.0019379017,0.0007187404,0.016284766],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999424,0.00006715594,0.00003232374,0.00014055757,0.00021651365,0.000119602584],"domain_scores_gemma":[0.99953127,0.00012254185,0.00003428746,0.00015554116,0.00012230357,0.000034089873],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00051822886,0.0009224473,0.0011451659,0.00067697937,0.0004294962,0.0012443305,0.002182275,0.001238972,0.0067802495],"category_scores_gemma":[0.0018805724,0.00042890254,0.0009239783,0.00072496355,0.0005761648,0.0032630914,0.0021602497,0.0009507205,0.0057529868],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010206341,0.00066039455,0.0013624614,0.0008315203,0.00016947402,0.00083380484,0.00038957523,0.083416626,0.30613872,0.02300074,0.039610602,0.54256546],"study_design_scores_gemma":[0.000071294664,0.00018723265,0.00042092972,0.000022917995,0.000049138343,0.00040224503,0.000115094335,0.8987833,0.07817357,0.011923458,0.009787461,0.00006333096],"about_ca_topic_score_codex":0.0037908328,"about_ca_topic_score_gemma":0.0074157827,"teacher_disagreement_score":0.0067802495,"about_ca_system_score_codex":0.000611377,"about_ca_system_score_gemma":0.0007603217,"threshold_uncertainty_score":0.02268213},"labels":[],"label_agreement":null},{"id":"W4392904379","doi":"10.1109/icassp48485.2024.10447954","title":"Multimodal Transformer with a Low-Computational-Cost Guarantee","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"National Research Foundation of Korea","keywords":"Computer science; Transformer; Action recognition; Inference; FLOPS; Modalities; Artificial intelligence; Machine learning; Computational complexity theory; Algorithm; Engineering; Parallel computing","score_opus":0.006684249304437783,"score_gpt":0.2641704279134448,"score_spread":0.257486178609007,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392904379","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02037263,0.00061655964,0.9646521,0.0008176677,0.00007525387,0.00016301285,0.00039785574,0.00577422,0.0071307095],"genre_scores_gemma":[0.713101,0.0007596057,0.2683626,0.0008968705,0.00022439368,0.00058263884,0.001703334,0.0011430464,0.013226495],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9993042,0.00016943607,0.000030791343,0.00017446397,0.00020074016,0.000120460696],"domain_scores_gemma":[0.99828136,0.00090385793,0.00008687853,0.00042782366,0.00021179311,0.00008825416],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012815851,0.0016598026,0.001244704,0.00064345833,0.0005557678,0.0013075947,0.0028883365,0.0015221467,0.01520894],"category_scores_gemma":[0.006492257,0.0005548135,0.00094704865,0.0007487215,0.00097165664,0.0038387384,0.0034893188,0.0022979828,0.0035446386],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00081964256,0.00035467735,0.0013393351,0.000395993,0.000091406386,0.00028236734,0.00013885152,0.40943974,0.014274761,0.06837546,0.027575465,0.47691232],"study_design_scores_gemma":[0.000043723634,0.00005994138,0.0001223316,0.000010343303,0.000018371164,0.00006763509,0.000018938732,0.9660474,0.002233159,0.030005718,0.0013634712,0.000009018158],"about_ca_topic_score_codex":0.0035000825,"about_ca_topic_score_gemma":0.0064193676,"teacher_disagreement_score":0.01520894,"about_ca_system_score_codex":0.0013167997,"about_ca_system_score_gemma":0.0024272373,"threshold_uncertainty_score":0.050878942},"labels":[],"label_agreement":null},{"id":"W4393147243","doi":"10.1609/aaai.v38i13.29407","title":"XKD: Cross-Modal Knowledge Distillation with Domain Alignment for Video Representation Learning","year":2024,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Vector Institute","funders":"Mitacs","keywords":"Modal; Representation (politics); Computer science; Domain (mathematical analysis); Artificial intelligence; Computer vision; Mathematics; Chemistry","score_opus":0.05995314657169955,"score_gpt":0.36507794310485103,"score_spread":0.3051247965331515,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393147243","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015916096,0.00093017536,0.97052956,0.0003533414,0.00013293627,0.0001418611,0.0010674383,0.00872787,0.0022007625],"genre_scores_gemma":[0.38525066,0.0005932584,0.5918438,0.0011986949,0.0001951815,0.0006787919,0.01006352,0.0007805117,0.009395578],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986628,0.00040275053,0.000062086125,0.00050819153,0.00022820148,0.00013583791],"domain_scores_gemma":[0.99877495,0.0005141965,0.00009637752,0.00032988607,0.00020751261,0.000077114804],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001818627,0.0018662572,0.001091334,0.0010650291,0.0006220787,0.0011957188,0.00319856,0.0018446337,0.0043736775],"category_scores_gemma":[0.0044669886,0.00063089246,0.0011405108,0.0010036903,0.0010839521,0.0026284694,0.003897308,0.0037713856,0.002035519],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005185932,0.0005339151,0.0018154804,0.00049658486,0.00022334653,0.00019208025,0.0002796903,0.16648822,0.018939817,0.011162733,0.028098805,0.7712507],"study_design_scores_gemma":[0.00004939598,0.00011334307,0.00029964992,0.000029198738,0.000022772729,0.000068689485,0.000047796115,0.9728767,0.009805495,0.012187341,0.0044759964,0.000023606251],"about_ca_topic_score_codex":0.004392272,"about_ca_topic_score_gemma":0.0062828157,"teacher_disagreement_score":0.004392272,"about_ca_system_score_codex":0.00111014,"about_ca_system_score_gemma":0.0014451762,"threshold_uncertainty_score":0.014631391},"labels":[],"label_agreement":null},{"id":"W4393147811","doi":"10.1609/aaai.v38i10.29037","title":"Make Prompts Adaptable: Bayesian Modeling for Vision-Language Prompt Learning with Data-Dependent Prior","year":2024,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea","keywords":"Computer science; Bayesian probability; Artificial intelligence; Machine learning; Bayesian statistics; Natural language processing; Bayesian inference; Psychology; Cognitive psychology","score_opus":0.07165001741030415,"score_gpt":0.3458690247856473,"score_spread":0.27421900737534316,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393147811","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0075922906,0.00017209961,0.99014926,0.00029084398,0.00003621502,0.000054383127,0.00015327612,0.0010017194,0.00054990186],"genre_scores_gemma":[0.54220384,0.00052068563,0.44728947,0.0008158844,0.00017701756,0.00062525197,0.001221243,0.0004961975,0.0066503915],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991998,0.00029333454,0.00003316176,0.00024771402,0.00015196446,0.00007400225],"domain_scores_gemma":[0.9977621,0.0013244941,0.00016069548,0.00023295096,0.0003653506,0.00015438185],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022330328,0.0009788241,0.0011061444,0.0006613277,0.00047268454,0.0011905635,0.0027227378,0.0018425215,0.0028562231],"category_scores_gemma":[0.012201844,0.00094143476,0.0009253184,0.0006475602,0.0010558743,0.0029609194,0.0019095769,0.003446839,0.00086974143],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036764695,0.00024450792,0.0019963447,0.00018693147,0.00006337899,0.00015728784,0.00034765352,0.71359164,0.0056218607,0.059226736,0.008066742,0.21012917],"study_design_scores_gemma":[0.000013889093,0.000021595504,0.000101295824,0.0000075247267,0.0000039567376,0.000012643895,0.000008249413,0.97930086,0.00060703437,0.019287195,0.0006269434,0.000008844741],"about_ca_topic_score_codex":0.007575768,"about_ca_topic_score_gemma":0.010675109,"teacher_disagreement_score":0.007575768,"about_ca_system_score_codex":0.001476009,"about_ca_system_score_gemma":0.0019058607,"threshold_uncertainty_score":0.015063345},"labels":[],"label_agreement":null},{"id":"W4393152659","doi":"10.1609/aaai.v38i17.29906","title":"YTCommentQA: Video Question Answerability in Instructional Videos","year":2024,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Computer science","score_opus":0.047121977550224924,"score_gpt":0.32778641041260964,"score_spread":0.28066443286238474,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393152659","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08553121,0.0037235513,0.011958641,0.0012475316,0.0005489584,0.001983584,0.85528207,0.024999566,0.014724994],"genre_scores_gemma":[0.04964767,0.00037476866,0.016679522,0.00038216685,0.0001004325,0.0012296186,0.92619884,0.00040564546,0.0049813422],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.997063,0.0007597123,0.00032355124,0.00081622216,0.0007003525,0.0003372708],"domain_scores_gemma":[0.99473476,0.002384537,0.0005126477,0.0007386331,0.0011093876,0.000520141],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015894156,0.003217053,0.0010257224,0.005011366,0.0012522772,0.0019093197,0.0027037004,0.0032104389,0.016862486],"category_scores_gemma":[0.012137567,0.0003864079,0.0016318371,0.0023123436,0.0007030268,0.003599055,0.0035228345,0.0023073754,0.011351906],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016663217,0.0009897347,0.01762756,0.004510047,0.00020980804,0.0007311465,0.0013864762,0.0042257747,0.00835432,0.002645267,0.84107804,0.11657549],"study_design_scores_gemma":[0.0011899401,0.0013217282,0.100534454,0.0015363831,0.00027153405,0.0019058487,0.004076476,0.086098254,0.023932118,0.009213117,0.7695485,0.0003716542],"about_ca_topic_score_codex":0.036733806,"about_ca_topic_score_gemma":0.078072794,"teacher_disagreement_score":0.036733806,"about_ca_system_score_codex":0.0026845527,"about_ca_system_score_gemma":0.0018005987,"threshold_uncertainty_score":0.07304001},"labels":[],"label_agreement":null},{"id":"W4393169473","doi":"10.1007/978-3-031-54845-1_5","title":"Empowering Canadian Geography Students and Faculty: New Approaches to Curriculum and Course Material Development Using Generative AI as a Writing Assistant","year":2024,"lang":"en","type":"book-chapter","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Course (navigation); Curriculum; Generative grammar; Mathematics education; Curriculum development; Pedagogy; Engineering ethics; Sociology; Engineering; Psychology; Computer science; Artificial intelligence","score_opus":0.06565042245054759,"score_gpt":0.3345357804165687,"score_spread":0.2688853579660211,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393169473","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07341802,0.002102263,0.048351195,0.04153224,0.0009721557,0.0002465533,0.00013187058,0.0011506351,0.832095],"genre_scores_gemma":[0.5224649,0.0023286287,0.06310612,0.0028205514,0.000087284025,0.00015466263,0.00016365007,0.00029678532,0.4085774],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99884665,0.00032090297,0.00001857158,0.0000999442,0.00042160545,0.00029229946],"domain_scores_gemma":[0.9973815,0.00058293476,0.00007689831,0.000101012214,0.00068185944,0.0011756939],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002160961,0.000412451,0.000134905,0.0009991385,0.007276608,0.0062563927,0.0015890017,0.0010606449,0.0089431815],"category_scores_gemma":[0.0039317883,0.00026135324,0.00018582844,0.0013716612,0.0039139367,0.0018223155,0.0029914733,0.0027290247,0.0010973613],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000043061296,0.00022997858,0.004120559,0.00012876344,0.000006637715,0.00043359937,0.1095318,0.0015634682,0.004924025,0.21444185,0.17980313,0.48477316],"study_design_scores_gemma":[0.000018174373,0.000042758456,0.00556287,0.00017966054,0.00001072283,0.00016912815,0.052925233,0.0016865631,0.0022760169,0.020548305,0.91653365,0.000046994624],"about_ca_topic_score_codex":0.75155956,"about_ca_topic_score_gemma":0.9485812,"teacher_disagreement_score":0.24844044,"about_ca_system_score_codex":0.024509313,"about_ca_system_score_gemma":0.07939555,"threshold_uncertainty_score":0.49980712},"labels":[],"label_agreement":null},{"id":"W4395689368","doi":"10.1038/s41598-024-60256-7","title":"A self-supervised framework for cross-modal search in histopathology archives using scale harmonization","year":2024,"lang":"en","type":"article","venue":"Scientific Reports","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Brock University; University of Waterloo","funders":"Mayo Clinic","keywords":"Harmonization; Computer science; Modal; Modalities; Digital pathology; Scale (ratio); Data science; Representation (politics); Artificial intelligence; Big data; Space (punctuation); Scarcity; Scheme (mathematics); Machine learning; Information retrieval; Pattern recognition (psychology); Data mining","score_opus":0.0269233348395202,"score_gpt":0.3435729190752458,"score_spread":0.3166495842357256,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4395689368","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015106339,0.00040967137,0.98000056,0.00021296018,0.00004942209,0.00011696306,0.00020232219,0.002680198,0.0012214838],"genre_scores_gemma":[0.40168598,0.00039524114,0.5814935,0.0009277163,0.000347655,0.00060606207,0.0030599474,0.0010103183,0.010473522],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99811244,0.00041109536,0.00012051731,0.0007113621,0.00038558402,0.00025898614],"domain_scores_gemma":[0.9983883,0.0005408828,0.00017236506,0.00037459927,0.00039590537,0.00012800327],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027797073,0.0011783628,0.0025841086,0.0024206592,0.0011715736,0.0017556446,0.004673666,0.0025971297,0.0042685363],"category_scores_gemma":[0.0039411066,0.00086471497,0.0019026208,0.0026927833,0.001234163,0.00246594,0.0036889552,0.0018532482,0.0019168155],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004931807,0.0005393431,0.0021686207,0.00025980314,0.000318344,0.00031580878,0.00047737962,0.34280995,0.012733875,0.013526557,0.017647507,0.60870963],"study_design_scores_gemma":[0.000018112434,0.000050595096,0.00021516692,0.00000749563,0.000016765409,0.000059784747,0.000052378087,0.99061453,0.0011992335,0.0066596577,0.0010921994,0.000014164064],"about_ca_topic_score_codex":0.006828765,"about_ca_topic_score_gemma":0.009734178,"teacher_disagreement_score":0.006828765,"about_ca_system_score_codex":0.0011162093,"about_ca_system_score_gemma":0.001928786,"threshold_uncertainty_score":0.014700651},"labels":[],"label_agreement":null},{"id":"W4395955659","doi":"10.55041/ijsrem31987","title":"IMAGE CAPTION GENERATOR USING DEEP LEARNING","year":2024,"lang":"en","type":"article","venue":"INTERANTIONAL JOURNAL OF SCIENTIFIC RESEARCH IN ENGINEERING AND MANAGEMENT","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Convolutional neural network; Closed captioning; Artificial intelligence; Encoder; Generator (circuit theory); Deep learning; Image (mathematics); Convolution (computer science); Task (project management); Pattern recognition (psychology); Computer vision; Artificial neural network","score_opus":0.039044449502799415,"score_gpt":0.35767770657940445,"score_spread":0.31863325707660506,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4395955659","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017193217,0.00089463417,0.92732686,0.00050018635,0.0007957254,0.0006252084,0.0023318282,0.0391856,0.011146714],"genre_scores_gemma":[0.23695017,0.0009718136,0.7297882,0.0004889359,0.00029273177,0.00069522,0.009328623,0.0017607965,0.019723423],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9996872,0.00006001684,0.000021229524,0.00012078807,0.00007708791,0.000033750588],"domain_scores_gemma":[0.9992021,0.00026221227,0.000062249404,0.00015940376,0.00027068032,0.000043456177],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00057221134,0.00131131,0.000589665,0.0008294485,0.00034777226,0.0010338752,0.0014564896,0.0011816494,0.013519634],"category_scores_gemma":[0.0023462465,0.0004061282,0.00095963903,0.0007539418,0.0004414669,0.0018354285,0.00092046184,0.0014455126,0.0058436883],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035351608,0.00022266635,0.00079741335,0.000851903,0.00012481982,0.0007178989,0.00024547387,0.07218205,0.066626824,0.01021371,0.06086931,0.7867944],"study_design_scores_gemma":[0.000034565775,0.00016815396,0.0005355217,0.00006956305,0.00004765507,0.00040870046,0.00009107025,0.86420846,0.091887616,0.012411751,0.03009004,0.000046933525],"about_ca_topic_score_codex":0.0018655928,"about_ca_topic_score_gemma":0.0018323027,"teacher_disagreement_score":0.013519634,"about_ca_system_score_codex":0.0008532656,"about_ca_system_score_gemma":0.00059169537,"threshold_uncertainty_score":0.045227706},"labels":[],"label_agreement":null},{"id":"W4396647722","doi":"10.2196/56627","title":"Advancing Accuracy in Multimodal Medical Tasks Through Bootstrapped Language-Image Pretraining (BioMedBLIP): Performance Evaluation Study","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Preprint; Computer science; Artificial intelligence; Natural language processing; Computer vision; World Wide Web","score_opus":0.020379232693986376,"score_gpt":0.38706889653301435,"score_spread":0.36668966383902796,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396647722","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7161466,0.020996653,0.18314062,0.004007955,0.0018097507,0.0016746382,0.007297874,0.047182374,0.017743476],"genre_scores_gemma":[0.8590098,0.0018641988,0.11422963,0.0019302289,0.00021565823,0.0006329332,0.01471207,0.0007341878,0.0066713006],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99810517,0.00065713545,0.0001393197,0.00057483005,0.0002901845,0.0002333453],"domain_scores_gemma":[0.9954059,0.0026643532,0.00021372561,0.0005279987,0.0009022898,0.00028575817],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0049679624,0.0028926549,0.0011797858,0.0012504249,0.0005762733,0.0011591204,0.002607274,0.0029357974,0.0036732664],"category_scores_gemma":[0.009845714,0.00064852776,0.0012188717,0.0008177438,0.00069374655,0.0017776723,0.0019746302,0.003744417,0.002423195],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00320464,0.0020644462,0.016648205,0.0012652448,0.00080321985,0.0005043304,0.00037881403,0.2395648,0.016921457,0.00078080484,0.040100235,0.67776376],"study_design_scores_gemma":[0.00014493306,0.0012712907,0.004976154,0.0001230553,0.00020056497,0.00032790098,0.00015548588,0.96892613,0.019700693,0.0007592342,0.003347219,0.00006745859],"about_ca_topic_score_codex":0.017482836,"about_ca_topic_score_gemma":0.017144,"teacher_disagreement_score":0.995032,"about_ca_system_score_codex":0.0019859504,"about_ca_system_score_gemma":0.0018182201,"threshold_uncertainty_score":0.034762144},"labels":[],"label_agreement":null},{"id":"W4396674347","doi":"10.32920/25761540","title":"An Attempt at Defining And Quantifying Image Describability Through Semantic Connection Between Visual and Language","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Closed captioning; Computer science; Natural language processing; Image (mathematics); Artificial intelligence; Task (project management); Ground truth; Judgement; Semantics (computer science); Domain (mathematical analysis); Mathematics; Programming language","score_opus":0.04252563624348574,"score_gpt":0.37419961974821603,"score_spread":0.3316739835047303,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396674347","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09605508,0.0008031145,0.88233876,0.001818152,0.00008518106,0.00015573986,0.00034936095,0.000537869,0.01785684],"genre_scores_gemma":[0.7986875,0.0007174886,0.19709639,0.00029724513,0.00009824191,0.00020273142,0.00049409736,0.00020662195,0.0021996868],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99742186,0.0010608152,0.00017279093,0.0007152101,0.00050032814,0.00012907213],"domain_scores_gemma":[0.98485017,0.00834113,0.0021078559,0.0028645592,0.0015373444,0.00029893557],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042016814,0.00078282383,0.0003909147,0.0021503998,0.0007604264,0.005202019,0.0013194449,0.0017027962,0.0034614329],"category_scores_gemma":[0.023376029,0.0005713938,0.0007131039,0.0012697187,0.006164138,0.011547749,0.0033347984,0.0026830288,0.00053332286],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038678694,0.000121860125,0.00729556,0.00091041904,0.00011963096,0.00034772637,0.0072851726,0.017390719,0.07161374,0.73055345,0.0024405587,0.16153434],"study_design_scores_gemma":[0.000027827862,0.00028363743,0.009802838,0.00032709952,0.0001175144,0.0009292135,0.003858689,0.17980379,0.067619294,0.7140769,0.022988815,0.0001644595],"about_ca_topic_score_codex":0.0012750839,"about_ca_topic_score_gemma":0.0007873549,"teacher_disagreement_score":0.005202019,"about_ca_system_score_codex":0.001379735,"about_ca_system_score_gemma":0.00062461436,"threshold_uncertainty_score":0.02222085},"labels":[],"label_agreement":null},{"id":"W4396674455","doi":"10.32920/25761540.v1","title":"An Attempt at Defining And Quantifying Image Describability Through Semantic Connection Between Visual and Language","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Closed captioning; Computer science; Image (mathematics); Natural language processing; Task (project management); Artificial intelligence; Judgement; Ground truth; Semantics (computer science); Domain (mathematical analysis); Mathematics; Programming language","score_opus":0.04252563624348574,"score_gpt":0.37419961974821603,"score_spread":0.3316739835047303,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396674455","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09605508,0.0008031145,0.88233876,0.001818152,0.00008518106,0.00015573986,0.00034936095,0.000537869,0.01785684],"genre_scores_gemma":[0.7986875,0.0007174886,0.19709639,0.00029724513,0.00009824191,0.00020273142,0.00049409736,0.00020662195,0.0021996868],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99742186,0.0010608152,0.00017279093,0.0007152101,0.00050032814,0.00012907213],"domain_scores_gemma":[0.98485017,0.00834113,0.0021078559,0.0028645592,0.0015373444,0.00029893557],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042016814,0.00078282383,0.0003909147,0.0021503998,0.0007604264,0.005202019,0.0013194449,0.0017027962,0.0034614329],"category_scores_gemma":[0.023376029,0.0005713938,0.0007131039,0.0012697187,0.006164138,0.011547749,0.0033347984,0.0026830288,0.00053332286],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038678694,0.000121860125,0.00729556,0.00091041904,0.00011963096,0.00034772637,0.0072851726,0.017390719,0.07161374,0.73055345,0.0024405587,0.16153434],"study_design_scores_gemma":[0.000027827862,0.00028363743,0.009802838,0.00032709952,0.0001175144,0.0009292135,0.003858689,0.17980379,0.067619294,0.7140769,0.022988815,0.0001644595],"about_ca_topic_score_codex":0.0012750839,"about_ca_topic_score_gemma":0.0007873549,"teacher_disagreement_score":0.005202019,"about_ca_system_score_codex":0.001379735,"about_ca_system_score_gemma":0.00062461436,"threshold_uncertainty_score":0.02222085},"labels":[],"label_agreement":null},{"id":"W4396723194","doi":"10.1145/3589334.3645603","title":"Multimodal Relation Extraction via a Mixture of Hierarchical Visual Context Learners","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Universitas Brawijaya","keywords":"Computer science; Relation (database); Relationship extraction; Context (archaeology); Extraction (chemistry); Artificial intelligence; Context model; Natural language processing; Data mining; Chromatography; Chemistry","score_opus":0.009990317320447086,"score_gpt":0.3100381747683123,"score_spread":0.3000478574478652,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396723194","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02545635,0.0012747294,0.964718,0.00031617098,0.00007494666,0.00016921286,0.0003672201,0.0038761527,0.0037472101],"genre_scores_gemma":[0.4343209,0.00092775165,0.555352,0.00066114217,0.00018424669,0.00028277698,0.0016523637,0.0003633318,0.0062554926],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99856514,0.0003132082,0.000067650355,0.00054157054,0.00035870206,0.0001538201],"domain_scores_gemma":[0.9989064,0.00045912495,0.00006944579,0.00021670408,0.0002613357,0.00008695017],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017006642,0.0017873046,0.0014547949,0.0025537675,0.0007608178,0.0017771348,0.0021206783,0.0016153274,0.0045466316],"category_scores_gemma":[0.005013397,0.0006240416,0.0024331815,0.0015976902,0.00073533354,0.0044081733,0.004395983,0.0022924882,0.002127293],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00062201894,0.00029134416,0.0021736159,0.00022914037,0.00020696565,0.00032123132,0.000557424,0.036384843,0.031616732,0.01307464,0.008985516,0.90553653],"study_design_scores_gemma":[0.00006566758,0.00020629243,0.0013214677,0.000075058255,0.00016583051,0.0003729554,0.0003265132,0.91811717,0.024330625,0.04762085,0.0073388726,0.000058665075],"about_ca_topic_score_codex":0.00439321,"about_ca_topic_score_gemma":0.0073938714,"teacher_disagreement_score":0.0045466316,"about_ca_system_score_codex":0.0007307052,"about_ca_system_score_gemma":0.00100539,"threshold_uncertainty_score":0.015210032},"labels":[],"label_agreement":null},{"id":"W4396832936","doi":"10.1145/3613905.3636316","title":"Computational Methodologies for Understanding, Automating, and Evaluating User Interfaces","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Engineering and Physical Sciences Research Council","keywords":"Computer science; Domain (mathematical analysis); Human–computer interaction; Field (mathematics); User interface; Data science; Interface (matter); Modeling language; Software engineering; Programming language; Software","score_opus":0.22434966393392727,"score_gpt":0.4633460451970769,"score_spread":0.23899638126314962,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396832936","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018644714,0.0006493036,0.9946727,0.00045965434,0.000024696295,0.00017509382,0.00011105341,0.0004064866,0.0016365576],"genre_scores_gemma":[0.051463224,0.00060182397,0.94609404,0.00013806683,0.000047347818,0.00075195864,0.0002864016,0.00014266517,0.00047449113],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9759615,0.013609428,0.0023741354,0.002766854,0.0047865156,0.0005015668],"domain_scores_gemma":[0.90948117,0.07315898,0.004796725,0.006991485,0.0050567267,0.00051479903],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01853577,0.0030049528,0.0027935754,0.0067409785,0.0014946439,0.009732258,0.005814125,0.0046952646,0.005872934],"category_scores_gemma":[0.10365768,0.0026021765,0.0034348476,0.005548462,0.0050220266,0.0085932575,0.004451495,0.0039958633,0.0013889683],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012301539,0.00033289057,0.0027704914,0.0015763714,0.0004577084,0.00014161416,0.0008754813,0.35389334,0.0024580236,0.30007002,0.004371185,0.33292982],"study_design_scores_gemma":[0.000039618568,0.000064408,0.0005810751,0.0001667821,0.00006847048,0.00008122786,0.00019492667,0.7389847,0.00139214,0.25384253,0.0045377435,0.00004630157],"about_ca_topic_score_codex":0.008769238,"about_ca_topic_score_gemma":0.008137988,"teacher_disagreement_score":0.01853577,"about_ca_system_score_codex":0.005067585,"about_ca_system_score_gemma":0.0049925493,"threshold_uncertainty_score":0.098027706},"labels":[],"label_agreement":null},{"id":"W4396893649","doi":"10.1111/cogs.13448","title":"Learning the Meanings of Function Words From Grounded Language Using a Visual Question Answering Model","year":2024,"lang":"en","type":"article","venue":"Cognitive Science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"Social Sciences and Humanities Research Council","keywords":"Function (biology); Question answering; Computer science; Meaning (existential); Artificial intelligence; Semantics (computer science); Visual reasoning; Natural language processing; Context (archaeology); Logical reasoning; Cognitive science; Linguistics; Psychology","score_opus":0.018542381159367582,"score_gpt":0.33580491063980183,"score_spread":0.31726252948043426,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396893649","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3573879,0.00016315238,0.6352383,0.00072345976,0.000039803326,0.000095414034,0.00030145192,0.0010397607,0.0050106742],"genre_scores_gemma":[0.8893181,0.00010227887,0.10680027,0.00015127692,0.000014532621,0.00011568248,0.00041698827,0.00007507657,0.0030057419],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996314,0.00013876891,0.000018943787,0.00013569945,0.000040146588,0.000034970424],"domain_scores_gemma":[0.998376,0.0010663355,0.00017532319,0.00017326775,0.00014345275,0.00006560678],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000918096,0.0006697173,0.00041350388,0.0004152823,0.0002102354,0.0011477809,0.0013909427,0.0010605692,0.0024234082],"category_scores_gemma":[0.005286507,0.00044289662,0.0012786746,0.00022265209,0.0009099828,0.0027188552,0.0009199699,0.0017110956,0.00040140795],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004676979,0.00033636403,0.011451637,0.0004517478,0.00026479736,0.0007835813,0.004423328,0.56151086,0.07044822,0.17223229,0.0030151566,0.17461434],"study_design_scores_gemma":[0.000021318006,0.0000696656,0.0007005882,0.000017170061,0.000027659573,0.00006191425,0.00012015219,0.9357633,0.004055622,0.058382966,0.00076190307,0.000017690178],"about_ca_topic_score_codex":0.0030898345,"about_ca_topic_score_gemma":0.0034333526,"teacher_disagreement_score":0.0030898345,"about_ca_system_score_codex":0.00087891717,"about_ca_system_score_gemma":0.0005812487,"threshold_uncertainty_score":0.008107126},"labels":[],"label_agreement":null},{"id":"W4400338820","doi":"10.1016/j.cviu.2024.104064","title":"Implicit and explicit commonsense for multi-sentence video captioning","year":2024,"lang":"en","type":"article","venue":"Computer Vision and Image Understanding","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; Canadian Institute for Advanced Research; University of British Columbia","funders":"Alliance de recherche numérique du Canada; Vector Institute; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Closed captioning; Computer science; Commonsense knowledge; Sentence; Paragraph; Natural language processing; Artificial intelligence; Task (project management); Object (grammar); Isolation (microbiology); Commonsense reasoning; Knowledge base; Transformer; Image (mathematics); World Wide Web","score_opus":0.07097240444014455,"score_gpt":0.34862258458640083,"score_spread":0.2776501801462563,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400338820","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032924786,0.0017884173,0.9154409,0.00063635723,0.00040863303,0.0005384969,0.0038111564,0.03538854,0.009062833],"genre_scores_gemma":[0.40912738,0.000998681,0.5627117,0.0005226327,0.00034036816,0.00052924815,0.014532796,0.0015237972,0.009713387],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99902546,0.00033752684,0.00006216277,0.0003127099,0.00018713316,0.000075067765],"domain_scores_gemma":[0.9971916,0.0012425816,0.0001868399,0.0007595343,0.0004818423,0.00013766924],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016307234,0.0018090904,0.00074861594,0.0011339146,0.0004730717,0.0015420993,0.0021168613,0.0016320926,0.0075288797],"category_scores_gemma":[0.0072543216,0.00040010616,0.0009170229,0.00071047473,0.00081887026,0.0032908893,0.0018386417,0.0022625162,0.0037366378],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078115443,0.00034838004,0.0011059802,0.0013213845,0.0001562288,0.0006508174,0.0006244618,0.095973656,0.099459186,0.01405281,0.04955063,0.73597527],"study_design_scores_gemma":[0.000055723143,0.00027637396,0.00076589047,0.00006843972,0.00006189615,0.0003507597,0.00014749225,0.9066022,0.056529086,0.014298984,0.020787526,0.00005566557],"about_ca_topic_score_codex":0.0022998892,"about_ca_topic_score_gemma":0.003830642,"teacher_disagreement_score":0.0075288797,"about_ca_system_score_codex":0.0008739121,"about_ca_system_score_gemma":0.00069117325,"threshold_uncertainty_score":0.025186598},"labels":[],"label_agreement":null},{"id":"W4400367343","doi":"10.7717/peerj-cs.2097","title":"Knowledge enhanced bottom-up affordance grounding for robotic interaction","year":2024,"lang":"en","type":"article","venue":"PeerJ Computer Science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Canadian Institute for Advanced Research","keywords":"Affordance; Computer science; Artificial intelligence; Ground; Natural language; Robot; Human–computer interaction; Top-down and bottom-up design; Natural language processing; Robotics; Machine learning; Programming language; Engineering","score_opus":0.029840325518459308,"score_gpt":0.35218747684773544,"score_spread":0.3223471513292761,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400367343","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06762283,0.0034243367,0.86475223,0.0008563924,0.00020323893,0.00035196033,0.014973,0.040119883,0.0076961503],"genre_scores_gemma":[0.38886824,0.0011228821,0.5437545,0.0005447406,0.00010469472,0.000712796,0.056046907,0.001376666,0.007468614],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999071,0.0001474918,0.00004200735,0.00045204491,0.00018366928,0.000103703605],"domain_scores_gemma":[0.999401,0.00016101946,0.00005686035,0.00026672758,0.000068052934,0.000046371788],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00056586135,0.0023844866,0.0012703036,0.0024611473,0.0008501099,0.0011998757,0.0032359078,0.0020594108,0.0071504586],"category_scores_gemma":[0.0026400778,0.0007081689,0.002183829,0.0018170452,0.001212105,0.003687632,0.0036789055,0.0019144209,0.0026637756],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055474445,0.00055466744,0.0041144826,0.0010507405,0.00026949347,0.00071564614,0.0005717748,0.08940997,0.054807357,0.013140429,0.048437476,0.78637314],"study_design_scores_gemma":[0.00007754356,0.00023018582,0.0059986347,0.00016782091,0.00009634837,0.0005641745,0.0003693837,0.84266245,0.03406424,0.071880616,0.04377751,0.00011108566],"about_ca_topic_score_codex":0.010343371,"about_ca_topic_score_gemma":0.023393393,"teacher_disagreement_score":0.010343371,"about_ca_system_score_codex":0.0012849186,"about_ca_system_score_gemma":0.0010262493,"threshold_uncertainty_score":0.023920715},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"simulation_or_modeling","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"simulation_or_modeling","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"medium"}],"label_agreement":"agree"},{"id":"W4400617959","doi":"10.1016/j.eswa.2024.124762","title":"Open-vocabulary object detection via debiased curriculum self-training","year":2024,"lang":"en","type":"article","venue":"Expert Systems with Applications","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Artificial intelligence; Vocabulary; Curriculum; Natural language processing; Machine learning; Object (grammar); Computer vision; Pattern recognition (psychology); Psychology; Linguistics; Pedagogy","score_opus":0.014389172244817729,"score_gpt":0.28630577759657566,"score_spread":0.27191660535175793,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400617959","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18481068,0.00085366593,0.78641194,0.0004135979,0.0004122044,0.00023846299,0.0006442942,0.0132221775,0.012992915],"genre_scores_gemma":[0.7321156,0.00024755576,0.24743183,0.00048774478,0.000099360426,0.00020688259,0.0023896703,0.00046704218,0.016554328],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99917597,0.000103358536,0.000039892562,0.00033001538,0.00017326456,0.0001774565],"domain_scores_gemma":[0.99870574,0.00036532176,0.00007700092,0.00028714578,0.00046575724,0.00009897785],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009284986,0.00085732865,0.00086829497,0.0010569476,0.0005872169,0.0007632126,0.00170871,0.0013353709,0.006709638],"category_scores_gemma":[0.0030350406,0.00031176014,0.0006242867,0.0008005789,0.000453295,0.001651496,0.0028077716,0.0014193146,0.004164573],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026500126,0.00024256227,0.0024810391,0.000081457736,0.000038236663,0.00008896677,0.00010679047,0.008190936,0.038053557,0.0014798356,0.005459833,0.94351184],"study_design_scores_gemma":[0.00006475227,0.00037428967,0.005421062,0.000047861624,0.00007638279,0.00035718715,0.00022814109,0.90031594,0.07820019,0.0057467753,0.009125718,0.000041726384],"about_ca_topic_score_codex":0.0040166215,"about_ca_topic_score_gemma":0.006255871,"teacher_disagreement_score":0.006709638,"about_ca_system_score_codex":0.00046147278,"about_ca_system_score_gemma":0.0011738647,"threshold_uncertainty_score":0.022445977},"labels":[],"label_agreement":null},{"id":"W4400676369","doi":"10.1007/978-3-031-65112-0_3","title":"Concept-Based Analysis of Neural Networks via Vision-Language Models","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Artificial neural network; Artificial intelligence; Natural language processing; Computer vision","score_opus":0.01025510262428448,"score_gpt":0.278116982880562,"score_spread":0.26786188025627755,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400676369","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014676695,0.0007978211,0.9818494,0.00021150256,0.000056319222,0.000024969753,0.000097789445,0.00040831338,0.0018772226],"genre_scores_gemma":[0.67393064,0.0014780717,0.31395406,0.00015962808,0.0001826363,0.00015168238,0.00065318117,0.00032644597,0.009163776],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99969125,0.00008290781,0.000013728374,0.00006972182,0.00010697242,0.00003539097],"domain_scores_gemma":[0.9990835,0.0005924526,0.00007264058,0.000050809955,0.00017573379,0.000024827017],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00079869153,0.00052010984,0.0005351923,0.0009870826,0.0002915115,0.0011540514,0.0011758202,0.0006426169,0.002880347],"category_scores_gemma":[0.0032314567,0.0003558421,0.00090087444,0.0008459799,0.00052416074,0.0017127066,0.0006414956,0.0012462029,0.0005593339],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013063257,0.000090951995,0.00090422924,0.00018776536,0.00011244641,0.00013422735,0.00013712978,0.57412523,0.01189974,0.14270306,0.0043954514,0.26517916],"study_design_scores_gemma":[0.0000016617562,0.000006239528,0.0001351265,0.000005282899,0.0000057196053,0.00001428125,0.0000067993415,0.9723985,0.00060360687,0.026502453,0.00031627872,0.0000040261416],"about_ca_topic_score_codex":0.005246643,"about_ca_topic_score_gemma":0.0046500554,"teacher_disagreement_score":0.005246643,"about_ca_system_score_codex":0.0011233744,"about_ca_system_score_gemma":0.00067788514,"threshold_uncertainty_score":0.010432184},"labels":[],"label_agreement":null},{"id":"W4400819487","doi":"10.1145/3658172","title":"S3: Speech, Script and Scene driven Head and Eye Animation","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Graphics","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University; University of Toronto","funders":"","keywords":"Computer science; Animation; Computer graphics (images); Head (geology); Computer facial animation; Artificial intelligence; Computer vision; Computer animation; Speech recognition","score_opus":0.019613478951137613,"score_gpt":0.2959048077076894,"score_spread":0.27629132875655177,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400819487","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032672476,0.0002930304,0.9059378,0.0002076302,0.00025945462,0.00032919066,0.001251092,0.047493093,0.011556187],"genre_scores_gemma":[0.2564483,0.00030014775,0.7183735,0.00023976534,0.0001000369,0.0003544822,0.0036017045,0.0056001423,0.01498195],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996891,0.000053540116,0.000013529086,0.00010171704,0.00011020368,0.000031854735],"domain_scores_gemma":[0.99961865,0.0001386359,0.000021404541,0.000087197666,0.00008152596,0.000052553758],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004691945,0.0011660197,0.0005170716,0.0006824252,0.00035372766,0.00089469575,0.0010699864,0.0009243213,0.013716067],"category_scores_gemma":[0.0019394747,0.00048417295,0.00087047357,0.0002848901,0.0005779311,0.0006150901,0.0017410129,0.0008056505,0.003838193],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010459647,0.00019660778,0.0024440542,0.00091895787,0.00021576509,0.00062572287,0.0014429026,0.06453171,0.21358915,0.012588072,0.045179993,0.65722114],"study_design_scores_gemma":[0.00011335766,0.00022907525,0.0014436647,0.00007543216,0.000054293927,0.000538167,0.00024151737,0.84386134,0.091587625,0.007203153,0.054589786,0.0000625184],"about_ca_topic_score_codex":0.002482056,"about_ca_topic_score_gemma":0.004816993,"teacher_disagreement_score":0.013716067,"about_ca_system_score_codex":0.00038534254,"about_ca_system_score_gemma":0.00047716766,"threshold_uncertainty_score":0.045884848},"labels":[],"label_agreement":null},{"id":"W4401042805","doi":"10.18653/v1/2024.findings-naacl.267","title":"Semantically-Prompted Language Models Improve Visual Descriptions","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"Computer science; Natural language processing; Artificial intelligence; Visual language; Programming language; Linguistics","score_opus":0.012386323627262873,"score_gpt":0.29864222577329397,"score_spread":0.2862559021460311,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401042805","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13057469,0.001935196,0.7884946,0.0023823814,0.0017074447,0.0004907749,0.00866506,0.03794383,0.027806053],"genre_scores_gemma":[0.81228995,0.0010130096,0.15192239,0.0013127984,0.00036216996,0.0002852508,0.011605487,0.0025411476,0.018667813],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988331,0.00041589473,0.000057797813,0.00040269995,0.00020774112,0.000082801576],"domain_scores_gemma":[0.9957918,0.0025837491,0.0002211984,0.00064657826,0.00060626026,0.00015050537],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010040073,0.0023816682,0.0008331291,0.0010408509,0.00049447245,0.00246296,0.0017826216,0.001958462,0.022771435],"category_scores_gemma":[0.010100285,0.0007049573,0.0014225513,0.00078569364,0.0004799876,0.0059946505,0.0024151788,0.0027964956,0.009521676],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0033939835,0.0015046626,0.0045195683,0.001348769,0.00036459762,0.00082851324,0.0006407966,0.17474908,0.06132601,0.031047337,0.077591054,0.64268565],"study_design_scores_gemma":[0.00012899097,0.00017382098,0.000608492,0.00007446582,0.00013226617,0.00013881297,0.00019768365,0.94825256,0.016277717,0.027372189,0.0065890504,0.00005392015],"about_ca_topic_score_codex":0.005711624,"about_ca_topic_score_gemma":0.007082063,"teacher_disagreement_score":0.022771435,"about_ca_system_score_codex":0.001056404,"about_ca_system_score_gemma":0.0013358352,"threshold_uncertainty_score":0.076178074},"labels":[],"label_agreement":null},{"id":"W4401413869","doi":"10.1109/icra57147.2024.10610980","title":"Aligning Knowledge Graph with Visual Perception for Object-goal Navigation","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"National Natural Science Foundation of China","keywords":"Computer science; Perception; Graph; Artificial intelligence; Goal orientation; Object (grammar); Human–computer interaction; Computer vision; Psychology; Theoretical computer science; Neuroscience; Social psychology","score_opus":0.01089643671907688,"score_gpt":0.32150315258729995,"score_spread":0.3106067158682231,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401413869","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020101273,0.000103754006,0.9745283,0.00009063733,0.000024409414,0.000041145326,0.00013298122,0.0037180677,0.0012594197],"genre_scores_gemma":[0.5299376,0.00015026091,0.46599442,0.0001617886,0.000019786592,0.00010766955,0.00082737976,0.0005628597,0.0022381456],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996524,0.00007398905,0.000012028928,0.00013814105,0.00008550103,0.00003803664],"domain_scores_gemma":[0.9995708,0.00015714615,0.00005198172,0.0000948467,0.000087135304,0.000038191207],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00036852006,0.00077819754,0.00048752403,0.0010107037,0.0003657225,0.0006081049,0.0012254169,0.00084026303,0.0024544834],"category_scores_gemma":[0.0022981907,0.000330968,0.0006960917,0.00089649035,0.0005712441,0.0019018729,0.0014776046,0.0009765718,0.0007337854],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002787875,0.0002330139,0.0023236559,0.00022372062,0.00009619801,0.0002706728,0.0004951498,0.2858797,0.04971047,0.016230503,0.006542903,0.63771516],"study_design_scores_gemma":[0.00001379876,0.00006854861,0.00065020897,0.000007928391,0.000018170262,0.00006383586,0.00008072549,0.9736222,0.008838313,0.014596436,0.0020226105,0.000017289969],"about_ca_topic_score_codex":0.009986022,"about_ca_topic_score_gemma":0.01280763,"teacher_disagreement_score":0.009986022,"about_ca_system_score_codex":0.00062478933,"about_ca_system_score_gemma":0.0010023343,"threshold_uncertainty_score":0.019855797},"labels":[],"label_agreement":null},{"id":"W4401416408","doi":"10.1109/icra57147.2024.10611485","title":"Talk2BEV: Language-enhanced Bird’s-eye View Maps for Autonomous Driving","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Artificial intelligence; Human–computer interaction; Computer vision","score_opus":0.008734051775047096,"score_gpt":0.30360339631739724,"score_spread":0.29486934454235014,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401416408","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05977913,0.0016508616,0.4641628,0.0012700339,0.00072171754,0.0010067272,0.04567914,0.40595162,0.01977794],"genre_scores_gemma":[0.35419136,0.0007079278,0.44729152,0.0016263592,0.00013301786,0.0017413718,0.1623005,0.01025917,0.021748837],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99917907,0.00023902356,0.000032747488,0.00023270697,0.00021810298,0.000098364864],"domain_scores_gemma":[0.99918836,0.00035469982,0.000028842833,0.00015468571,0.00019895424,0.00007450198],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009871621,0.0022621762,0.00067883777,0.0009018858,0.00045066065,0.0015984061,0.0026358725,0.0019023799,0.014928268],"category_scores_gemma":[0.004110025,0.00062765984,0.0013814943,0.0004724816,0.00040307423,0.0032366833,0.0032325212,0.0016219482,0.008278391],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014187278,0.0005967134,0.0033998166,0.0013046054,0.0003374622,0.00065609097,0.0013508726,0.08167995,0.02963379,0.00852564,0.41232526,0.45877096],"study_design_scores_gemma":[0.00022153281,0.0003767428,0.0023479885,0.00015768979,0.000078825535,0.00031787087,0.0008923773,0.7315083,0.029364452,0.019303355,0.215273,0.00015781616],"about_ca_topic_score_codex":0.015060429,"about_ca_topic_score_gemma":0.020223644,"teacher_disagreement_score":0.015060429,"about_ca_system_score_codex":0.001018174,"about_ca_system_score_gemma":0.0010078816,"threshold_uncertainty_score":0.04994005},"labels":[],"label_agreement":null},{"id":"W4401690668","doi":"10.1016/j.heliyon.2024.e36272","title":"Enhancing image caption generation through context-aware attention mechanism","year":2024,"lang":"en","type":"article","venue":"Heliyon","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Athabasca University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Context (archaeology); Mechanism (biology); Image (mathematics); Psychology; Computer science; Computer vision; History; Epistemology; Philosophy","score_opus":0.021577888728825483,"score_gpt":0.29426680512841585,"score_spread":0.27268891639959036,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401690668","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.083160445,0.013666031,0.7109622,0.0017624837,0.0050597577,0.0013160547,0.007460592,0.12868825,0.04792414],"genre_scores_gemma":[0.4063702,0.004580615,0.52857184,0.0014410575,0.0012597755,0.0005377126,0.022010595,0.0022723952,0.032955874],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993759,0.000120101104,0.000034709436,0.00024336124,0.00015881477,0.00006726097],"domain_scores_gemma":[0.9990501,0.00022890407,0.00006012136,0.00021761112,0.00037465713,0.000068610665],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00072025234,0.0023123755,0.000816183,0.0015574329,0.0006803553,0.001388122,0.0013583603,0.0013810621,0.0093676],"category_scores_gemma":[0.0035233665,0.00033592212,0.0007659989,0.0011451872,0.00034755978,0.0019643137,0.0012383242,0.001500392,0.006926108],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00075440167,0.0003535385,0.0009702428,0.0009154393,0.00022223494,0.00056699396,0.00013254481,0.021801133,0.15016171,0.0019643528,0.09865199,0.7235054],"study_design_scores_gemma":[0.00015116372,0.00042460184,0.0034083019,0.0000941789,0.00025490095,0.00094442384,0.00014429564,0.6584399,0.2781099,0.00279672,0.05507594,0.00015566737],"about_ca_topic_score_codex":0.0073092794,"about_ca_topic_score_gemma":0.0115260305,"teacher_disagreement_score":0.0093676,"about_ca_system_score_codex":0.000698689,"about_ca_system_score_gemma":0.00086990104,"threshold_uncertainty_score":0.031337798},"labels":[],"label_agreement":null},{"id":"W4402593760","doi":"10.1109/iciea61579.2024.10665031","title":"TALON: Improving Large Language Model Cognition with Tactility-Vision Fusion","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Computer science; Cognition; Artificial intelligence; Psychology; Neuroscience","score_opus":0.006413338553997826,"score_gpt":0.28071533987002073,"score_spread":0.2743020013160229,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402593760","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.110470854,0.0024935338,0.8342378,0.0010207329,0.0005487192,0.00029265834,0.0030397614,0.039536845,0.008359109],"genre_scores_gemma":[0.7221697,0.0005754424,0.25243115,0.0019715698,0.00019957586,0.00041397606,0.011854196,0.0011041862,0.009280236],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9988827,0.00027017298,0.00005682512,0.0003904658,0.0002557156,0.00014415575],"domain_scores_gemma":[0.9988569,0.00049085426,0.000059948525,0.00027095067,0.00024954122,0.000071804156],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001809217,0.0028288432,0.0012071743,0.00099786,0.0005376493,0.0016964524,0.0022686278,0.0014984424,0.004939771],"category_scores_gemma":[0.004981258,0.00040759292,0.0019040357,0.00067215256,0.00053013524,0.0036958454,0.0027805017,0.002402781,0.002747967],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00064741325,0.0007153054,0.004146014,0.00042041892,0.00053258415,0.0004281336,0.00037587795,0.150296,0.028322954,0.005503025,0.035397656,0.77321464],"study_design_scores_gemma":[0.00003805605,0.00020921788,0.00072485243,0.000024977744,0.0000717385,0.00011747168,0.00009977643,0.9767785,0.010846626,0.0072007906,0.0038440733,0.0000439094],"about_ca_topic_score_codex":0.009008889,"about_ca_topic_score_gemma":0.013160297,"teacher_disagreement_score":0.009008889,"about_ca_system_score_codex":0.0010006803,"about_ca_system_score_gemma":0.0010885841,"threshold_uncertainty_score":0.017912865},"labels":[],"label_agreement":null},{"id":"W4402703105","doi":"10.1109/cvpr52733.2024.01295","title":"MM-Narrator: Narrating Long-form Videos with Multimodal In-Context Learning","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Advanced Micro Devices (Canada)","funders":"","keywords":"Context (archaeology); Computer science; Artificial intelligence; Human–computer interaction; Multimedia; History; Archaeology","score_opus":0.008162838442675452,"score_gpt":0.2595758568692203,"score_spread":0.25141301842654484,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402703105","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06899493,0.0015521194,0.8139687,0.00047827634,0.0005334139,0.0011246523,0.0035654462,0.0984881,0.011294284],"genre_scores_gemma":[0.28720948,0.00040720514,0.6914251,0.0006113468,0.00012456621,0.0007957953,0.0068572145,0.0013015798,0.011267817],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99948466,0.00015633396,0.000026781092,0.00020218754,0.000091142276,0.000038877803],"domain_scores_gemma":[0.99933714,0.0002934913,0.000037486287,0.00017426036,0.00009351685,0.00006406066],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00093691685,0.0013503317,0.0005547514,0.00047779645,0.00029809398,0.0008153699,0.0022485564,0.0011565275,0.01245899],"category_scores_gemma":[0.0046059615,0.0003342295,0.00058831036,0.00022184006,0.00033737082,0.0018728159,0.0020180945,0.00104559,0.003539302],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010359016,0.0005313728,0.0024009845,0.0012400376,0.00019453232,0.0010958379,0.000990526,0.025480418,0.06610407,0.004396525,0.05443955,0.8420902],"study_design_scores_gemma":[0.00025338164,0.001067141,0.0022776097,0.00015259672,0.00014514424,0.0014077539,0.000592152,0.8216229,0.09423789,0.008880176,0.06924021,0.00012306783],"about_ca_topic_score_codex":0.0010727196,"about_ca_topic_score_gemma":0.0020009761,"teacher_disagreement_score":0.01245899,"about_ca_system_score_codex":0.00035482904,"about_ca_system_score_gemma":0.0003953239,"threshold_uncertainty_score":0.041679442},"labels":[],"label_agreement":null},{"id":"W4402716477","doi":"10.1109/cvpr52733.2024.00913","title":"MMMU: A Massive Multi-Discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":245,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria; University of Waterloo","funders":"","keywords":"Benchmark (surveying); Computer science; Artificial intelligence; Cognitive science; Psychology; Geology","score_opus":0.043292078810416655,"score_gpt":0.3393487419959848,"score_spread":0.29605666318556817,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402716477","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.41260418,0.012002991,0.23671106,0.0054528425,0.002084343,0.0036511482,0.10358579,0.09968363,0.12422391],"genre_scores_gemma":[0.5216844,0.001554238,0.25897565,0.0019310754,0.0002845746,0.0023115012,0.19174995,0.0025995225,0.01890909],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.996995,0.0010619329,0.00023336045,0.0007736659,0.00063948054,0.00029660604],"domain_scores_gemma":[0.9946449,0.0028513155,0.00026794523,0.00085803977,0.0008578173,0.00051991397],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002941309,0.0028220692,0.00085919205,0.0028170939,0.0009989219,0.002621144,0.0037182977,0.0034432379,0.012264313],"category_scores_gemma":[0.016868725,0.00046314273,0.0016636852,0.00201437,0.00086998224,0.003952968,0.004265556,0.0027896385,0.0061685387],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022357763,0.0027639386,0.015611593,0.0033587269,0.0008044827,0.00072851614,0.0013385445,0.17284934,0.008460232,0.013225949,0.325524,0.4530988],"study_design_scores_gemma":[0.0006006227,0.001227517,0.014297461,0.00060330937,0.00023490505,0.0005195384,0.0015403944,0.78258884,0.01654877,0.030852059,0.15083039,0.0001562419],"about_ca_topic_score_codex":0.021649843,"about_ca_topic_score_gemma":0.026624436,"teacher_disagreement_score":0.021649843,"about_ca_system_score_codex":0.0027755958,"about_ca_system_score_gemma":0.0025575247,"threshold_uncertainty_score":0.043047667},"labels":[],"label_agreement":null},{"id":"W4402727410","doi":"10.1109/cvpr52733.2024.01307","title":"Contrasting Intra-Modal and Ranking Cross-Modal Hard Negatives to Enhance Visio-Linguistic Compositional Understanding","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Modal; Ranking (information retrieval); Computer science; Linguistics; Natural language processing; Artificial intelligence; Materials science; Philosophy","score_opus":0.024641684716629628,"score_gpt":0.3549871254119355,"score_spread":0.33034544069530586,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402727410","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.180974,0.0013168306,0.78092337,0.0012711365,0.00035491263,0.00040669827,0.0018659334,0.015362699,0.017524406],"genre_scores_gemma":[0.68268263,0.00033980655,0.29472783,0.0013319105,0.00021745257,0.0002986536,0.006505821,0.0013031436,0.012592836],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985977,0.00029176602,0.000060331062,0.0005365764,0.00030775194,0.00020585922],"domain_scores_gemma":[0.997241,0.0012962915,0.00014401933,0.00046103945,0.0006837735,0.00017391742],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025216604,0.002340433,0.001254423,0.0019921104,0.0009082344,0.002498183,0.0026188665,0.0023113305,0.0077649364],"category_scores_gemma":[0.007568032,0.00039596888,0.0012540043,0.0010453674,0.0010868525,0.003812132,0.003271962,0.0034065521,0.0029692377],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014272868,0.0008789004,0.0046057277,0.0005397721,0.00023356304,0.0005049806,0.000323168,0.086817354,0.052039195,0.013103611,0.030530667,0.8089957],"study_design_scores_gemma":[0.000051394036,0.00021161309,0.0012369897,0.00004281272,0.00006573027,0.00015719811,0.0001891692,0.9490126,0.026183445,0.019008769,0.0038068567,0.000033492175],"about_ca_topic_score_codex":0.0058758548,"about_ca_topic_score_gemma":0.012980542,"teacher_disagreement_score":0.0077649364,"about_ca_system_score_codex":0.0013787971,"about_ca_system_score_gemma":0.0015357337,"threshold_uncertainty_score":0.0259763},"labels":[],"label_agreement":null},{"id":"W4402728032","doi":"10.1109/cvpr52733.2024.00386","title":"Emergent Open-Vocabulary Semantic Segmentation from Off-the-Shelf Vision-Language Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of British Columbia","funders":"","keywords":"Computer science; Vocabulary; Segmentation; Natural language processing; Artificial intelligence; Image segmentation; Linguistics","score_opus":0.01812614014927617,"score_gpt":0.3248814039916032,"score_spread":0.30675526384232704,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402728032","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04624232,0.0019813867,0.90944886,0.00079155347,0.0003950806,0.00018952676,0.0020198652,0.03209128,0.006840118],"genre_scores_gemma":[0.6098004,0.0011574144,0.3563544,0.0016198276,0.00026446691,0.0004139573,0.012502336,0.0032713118,0.014615997],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99932575,0.000098012206,0.000027950584,0.00032702,0.00012142014,0.000099856625],"domain_scores_gemma":[0.99933726,0.0002434707,0.00005715045,0.0001587181,0.00013933706,0.00006399699],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00088938145,0.0023686804,0.0012804649,0.0012639307,0.00048161548,0.0019513088,0.0034781797,0.0021116994,0.005347282],"category_scores_gemma":[0.002895748,0.0010446969,0.002040541,0.0009937155,0.0010791857,0.004510208,0.0022707896,0.002744787,0.0049966844],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00068319455,0.00044035097,0.0020434747,0.0007580991,0.00034291542,0.0004823288,0.0005267244,0.27922267,0.06330973,0.020806104,0.037765127,0.59361935],"study_design_scores_gemma":[0.000031696145,0.000070844464,0.00025025112,0.000032943317,0.000035943507,0.00008752346,0.00006790245,0.96714884,0.009813645,0.018064994,0.004374817,0.000020637548],"about_ca_topic_score_codex":0.010634489,"about_ca_topic_score_gemma":0.018911414,"teacher_disagreement_score":0.010634489,"about_ca_system_score_codex":0.0015656174,"about_ca_system_score_gemma":0.0015950274,"threshold_uncertainty_score":0.021145165},"labels":[],"label_agreement":null},{"id":"W4402728164","doi":"10.1109/cvpr52733.2024.01335","title":"Jack of All Tasks, Master of Many: Designing General-purpose Coarse-to-Fine Vision-Language Model","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Human–computer interaction; Artificial intelligence; Natural language processing; Computer graphics (images); Programming language; Computer vision","score_opus":0.026130957254922278,"score_gpt":0.32129488570047404,"score_spread":0.2951639284455518,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402728164","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04834696,0.0014407642,0.903467,0.0015114661,0.00027413442,0.00030560986,0.00159757,0.03503687,0.008019627],"genre_scores_gemma":[0.44266704,0.0007834652,0.53558266,0.0016428048,0.000102071266,0.0004977681,0.0045479108,0.0020742437,0.012102114],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999276,0.00014076439,0.000039615723,0.00034877154,0.00009203389,0.00010275753],"domain_scores_gemma":[0.9990397,0.00026450396,0.0000614723,0.0004190918,0.000119639866,0.00009548579],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010847191,0.0013301692,0.0010314274,0.00047862288,0.00057240087,0.0023125734,0.003101812,0.001458654,0.0058246963],"category_scores_gemma":[0.0036628419,0.00065846456,0.0014889934,0.00051905646,0.001222589,0.0052333605,0.0022822048,0.0032792091,0.0036306079],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00086565426,0.0004463577,0.0040309774,0.0008508632,0.00021108877,0.00037228022,0.0007513478,0.24264601,0.04461756,0.060270216,0.056599654,0.58833796],"study_design_scores_gemma":[0.000038437487,0.00010278658,0.00032211226,0.000038346538,0.000040685638,0.00008432334,0.000095279065,0.93107986,0.013696143,0.044427115,0.010042139,0.000032656335],"about_ca_topic_score_codex":0.007726752,"about_ca_topic_score_gemma":0.01279715,"teacher_disagreement_score":0.007726752,"about_ca_system_score_codex":0.0014228441,"about_ca_system_score_gemma":0.0025710894,"threshold_uncertainty_score":0.019485533},"labels":[],"label_agreement":null},{"id":"W4402753993","doi":"10.1109/cvpr52733.2024.02677","title":"SemCity: Semantic Scene Generation with Triplane Diffusion","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Ministry of Science and ICT, South Korea; Iran Telecommunication Research Center","keywords":"Computer science; Diffusion; Artificial intelligence; Computer vision; Physics","score_opus":0.014081222439961093,"score_gpt":0.2562263197521697,"score_spread":0.2421450973122086,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402753993","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008911707,0.00033397906,0.9711175,0.00025495994,0.00010792492,0.00017506603,0.0010994052,0.015457629,0.0025418268],"genre_scores_gemma":[0.18453674,0.00048481036,0.7964619,0.0004291619,0.00006304137,0.0003765304,0.0063582086,0.0044216197,0.0068679657],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995602,0.00005698407,0.00001738622,0.00016568632,0.0001549447,0.000044876724],"domain_scores_gemma":[0.99947375,0.00016192113,0.000037826343,0.00018404779,0.00008995849,0.00005258414],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00070210017,0.0015498667,0.00081022014,0.0012372367,0.00047536354,0.0014075777,0.002264908,0.0014063208,0.0066564186],"category_scores_gemma":[0.0021658475,0.00076223584,0.0019509846,0.00083930197,0.0007933623,0.0017885665,0.0022713428,0.002243707,0.002539399],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004046071,0.00025688755,0.0019033654,0.00052279315,0.00023194127,0.0003644728,0.0005330111,0.53977007,0.046267632,0.02415671,0.040938657,0.3446498],"study_design_scores_gemma":[0.0000438405,0.0000498914,0.00023174181,0.00001694364,0.000014887131,0.00013807538,0.000039861614,0.97011894,0.010675841,0.0072921147,0.011349649,0.000028114731],"about_ca_topic_score_codex":0.0071384856,"about_ca_topic_score_gemma":0.014014948,"teacher_disagreement_score":0.0071384856,"about_ca_system_score_codex":0.0011299911,"about_ca_system_score_gemma":0.0007116789,"threshold_uncertainty_score":0.022267878},"labels":[],"label_agreement":null},{"id":"W4402754270","doi":"10.1109/cvpr52733.2024.02674","title":"LLM4SGG: Large Language Models for Weakly Supervised Scene Graph Generation","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing; Graph; Language model; Theoretical computer science","score_opus":0.02582463357507845,"score_gpt":0.30266158494447193,"score_spread":0.27683695136939346,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402754270","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008552405,0.00035694797,0.9563112,0.00052856427,0.00012204301,0.0003035237,0.0017783755,0.03040858,0.0016383715],"genre_scores_gemma":[0.2023904,0.00028146862,0.76997334,0.001330324,0.00013589764,0.0012741098,0.013506365,0.0032942793,0.0078138355],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987457,0.00047545962,0.000055077537,0.0004436272,0.00017970512,0.0001004356],"domain_scores_gemma":[0.9981198,0.0009707123,0.0000948256,0.00046102036,0.0002446591,0.0001090094],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018986207,0.0023424223,0.0014090054,0.001161469,0.0009387973,0.0016062357,0.0058200397,0.0027590508,0.008392759],"category_scores_gemma":[0.0058966232,0.0012261191,0.0027501162,0.0008808513,0.0012049358,0.0031351382,0.0033312866,0.0043760166,0.0042923717],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00057610293,0.00044061386,0.0021548427,0.0004587133,0.00025017397,0.00055771956,0.00056699687,0.49263126,0.011059315,0.026562614,0.059143938,0.4055978],"study_design_scores_gemma":[0.000019342699,0.00001501835,0.000058126046,0.0000058733617,0.0000063136126,0.000017966866,0.0000126779105,0.9878,0.0010216018,0.008808224,0.0022264754,0.000008369291],"about_ca_topic_score_codex":0.015209776,"about_ca_topic_score_gemma":0.030621286,"teacher_disagreement_score":0.015209776,"about_ca_system_score_codex":0.0022651388,"about_ca_system_score_gemma":0.0019094242,"threshold_uncertainty_score":0.030242443},"labels":[],"label_agreement":null},{"id":"W4402772286","doi":"10.1109/cvpr52733.2024.01355","title":"EgoThink: Evaluating First-Person Perspective Thinking Capability of Vision-Language Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Perspective (graphical); Computer science; Human–computer interaction; Artificial intelligence; Natural language processing","score_opus":0.02926812550214529,"score_gpt":0.357759155973917,"score_spread":0.3284910304717717,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402772286","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8472778,0.004528296,0.11348786,0.0008199697,0.00047460682,0.00094871724,0.0049605,0.009348582,0.018153679],"genre_scores_gemma":[0.93131834,0.0004763317,0.057303984,0.00028334328,0.00006540399,0.00039944318,0.007576407,0.00021476395,0.0023619742],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9972851,0.0013547142,0.00018547136,0.0006053014,0.00039238308,0.000176923],"domain_scores_gemma":[0.9927965,0.005033793,0.0004497156,0.00072426983,0.00056734367,0.0004284132],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043906914,0.0018124896,0.0006694018,0.0014458783,0.0004680025,0.0015074528,0.001561127,0.0017062097,0.0036872695],"category_scores_gemma":[0.016336871,0.00023636624,0.00080232596,0.00066259294,0.00060468115,0.002536607,0.002290382,0.0016158867,0.0011112065],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0037286845,0.0026819517,0.04813377,0.00313733,0.0013202482,0.0004887738,0.0026888328,0.18022075,0.016712435,0.0064300606,0.03594643,0.69851065],"study_design_scores_gemma":[0.00028939324,0.0023429345,0.019641265,0.0002082216,0.00026762893,0.00025767757,0.0012485047,0.9405463,0.016018754,0.0094231535,0.009647614,0.000108602806],"about_ca_topic_score_codex":0.004265076,"about_ca_topic_score_gemma":0.0058361306,"teacher_disagreement_score":0.0043906914,"about_ca_system_score_codex":0.0010466275,"about_ca_system_score_gemma":0.0008887474,"threshold_uncertainty_score":0.02322042},"labels":[],"label_agreement":null},{"id":"W4402904107","doi":"10.1109/cvprw63382.2024.00708","title":"TrafficVLM: A Controllable Visual Language Model for Traffic Video Captioning","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Closed captioning; Computer science; Speech recognition; Artificial intelligence; Image (mathematics)","score_opus":0.012493589958756626,"score_gpt":0.31336517459600666,"score_spread":0.30087158463725006,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402904107","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008736009,0.00024066216,0.9771871,0.00017446841,0.000084366606,0.00015111761,0.0013034024,0.010145382,0.0019775813],"genre_scores_gemma":[0.40203238,0.0005704785,0.5759635,0.0006375075,0.00012547296,0.0008915401,0.007392135,0.002261359,0.010125584],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99963975,0.000075882745,0.000016359352,0.00013302552,0.00009017324,0.000044743458],"domain_scores_gemma":[0.9995353,0.00017847195,0.00004091855,0.00007784647,0.00012415025,0.000043366654],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005875469,0.0010675128,0.00055915426,0.0008025985,0.0003054919,0.00096112996,0.0023813534,0.0010033882,0.0045043714],"category_scores_gemma":[0.0025191873,0.00053215853,0.0012121865,0.00055713695,0.0005254351,0.0015467942,0.0015368976,0.0015348288,0.0018818438],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004637541,0.00028284208,0.0017928035,0.0003512103,0.0001251689,0.00038228877,0.00029652606,0.5302443,0.029113498,0.02309627,0.037114628,0.3767367],"study_design_scores_gemma":[0.000009709384,0.000018573368,0.00009853641,0.0000065470135,0.0000063402417,0.000034067045,0.000009047614,0.9917761,0.0019218434,0.003597254,0.0025130727,0.000008955655],"about_ca_topic_score_codex":0.008989243,"about_ca_topic_score_gemma":0.013911239,"teacher_disagreement_score":0.008989243,"about_ca_system_score_codex":0.001020112,"about_ca_system_score_gemma":0.0008672308,"threshold_uncertainty_score":0.017873824},"labels":[],"label_agreement":null},{"id":"W4402961730","doi":"10.1007/978-3-031-72784-9_27","title":"SafaRi: Adaptive Sequence Transformer for Weakly Supervised Referring Expression Segmentation","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Segmentation; Transformer; Artificial intelligence; Sequence (biology); Computer vision; Pattern recognition (psychology); Electrical engineering; Voltage; Biology","score_opus":0.039102226118176285,"score_gpt":0.3031783288866046,"score_spread":0.26407610276842836,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402961730","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004279163,0.00029122684,0.9488826,0.000079760626,0.000119021504,0.00012509013,0.0012743383,0.042956054,0.0019927705],"genre_scores_gemma":[0.07103743,0.00038623333,0.8997331,0.0002852471,0.00013461738,0.00031322768,0.009225508,0.005969573,0.012915006],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989975,0.00017168434,0.000056885823,0.0003955756,0.00027445023,0.000103806786],"domain_scores_gemma":[0.9992797,0.00023773583,0.000039831542,0.0002113567,0.00018219427,0.000049336613],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009619746,0.0018446082,0.0018359806,0.0016159727,0.0006558088,0.0017371072,0.0029248402,0.0014934893,0.022097291],"category_scores_gemma":[0.0023470034,0.00087034336,0.0015923816,0.0018807556,0.000692039,0.0025751328,0.0025866823,0.0022830355,0.01750138],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00075060857,0.00013851917,0.0002661632,0.0003184015,0.00009347601,0.00020704925,0.00015973242,0.007872971,0.079108894,0.01121648,0.04827654,0.8515912],"study_design_scores_gemma":[0.000089135734,0.0002732516,0.00083816645,0.00005895075,0.000105093386,0.0007130719,0.00012681905,0.7844941,0.1381306,0.033815764,0.04126449,0.000090554786],"about_ca_topic_score_codex":0.0032553168,"about_ca_topic_score_gemma":0.006178863,"teacher_disagreement_score":0.022097291,"about_ca_system_score_codex":0.000722247,"about_ca_system_score_gemma":0.0010996289,"threshold_uncertainty_score":0.07392281},"labels":[],"label_agreement":null},{"id":"W4402963229","doi":"10.1145/3664647.3680815","title":"Text-Region Matching for Multi-Label Image Recognition with Missing Labels","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"RIKEN; Natural Science Foundation of Anhui Province; National Natural Science Foundation of China","keywords":"Computer science; Bridging (networking); Pascal (unit); Artificial intelligence; Matching (statistics); Semantics (computer science); Benchmark (surveying); Class (philosophy); Margin (machine learning); Pattern recognition (psychology); Image (mathematics); Visualization; Semantic gap; Natural language processing; Machine learning; Image retrieval; Mathematics","score_opus":0.07659082094066458,"score_gpt":0.34661633013744864,"score_spread":0.27002550919678403,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402963229","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017012231,0.00078207423,0.96349895,0.00024814185,0.00014782773,0.00018589396,0.00065506255,0.015283312,0.0021865012],"genre_scores_gemma":[0.22524196,0.00038243367,0.7589232,0.0007967235,0.00016910597,0.00046098334,0.0046460023,0.0018004904,0.0075790687],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9979818,0.00034137428,0.0000898656,0.0009781638,0.00039937845,0.00020943672],"domain_scores_gemma":[0.99795294,0.00055552326,0.00020251538,0.0007752894,0.0004056336,0.000108105276],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019996986,0.001608556,0.0016765696,0.0020335393,0.0008156656,0.0014391484,0.00457093,0.0025318663,0.007004326],"category_scores_gemma":[0.0056373384,0.0005902492,0.0018571253,0.0018281456,0.0012262412,0.003218587,0.0027236626,0.0023360427,0.00504892],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00088831154,0.0003838068,0.0012924327,0.0004845911,0.00012084648,0.00031266912,0.00026971864,0.05376077,0.06558403,0.008495124,0.023029912,0.8453778],"study_design_scores_gemma":[0.00006393231,0.00015685505,0.0007327238,0.00003998932,0.000053692125,0.00026934483,0.00013184646,0.9136119,0.051922027,0.023746219,0.009229021,0.0000424783],"about_ca_topic_score_codex":0.0041630287,"about_ca_topic_score_gemma":0.006217479,"teacher_disagreement_score":0.007004326,"about_ca_system_score_codex":0.0015500897,"about_ca_system_score_gemma":0.0013796674,"threshold_uncertainty_score":0.023431778},"labels":[],"label_agreement":null},{"id":"W4403289479","doi":"10.1016/j.knosys.2024.112610","title":"Vision-and-language navigation based on history-aware cross-modal feature fusion in indoor environment","year":2024,"lang":"en","type":"article","venue":"Knowledge-Based Systems","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Science, Technology and Innovation Commission of Shenzhen Municipality; National Natural Science Foundation of China","keywords":"Modal; Computer science; Feature (linguistics); Fusion; Artificial intelligence; Computer vision; Human–computer interaction; Linguistics; Materials science","score_opus":0.009174197934916771,"score_gpt":0.28593626638038167,"score_spread":0.2767620684454649,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403289479","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11417265,0.00079594227,0.8788373,0.00021741635,0.00022204989,0.00004193703,0.00029020204,0.0022948324,0.0031277817],"genre_scores_gemma":[0.86571664,0.00032267947,0.13028413,0.00014660026,0.00007134485,0.00004476363,0.0005281826,0.0001024423,0.0027832766],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99965763,0.00002932327,0.000014361053,0.0001299747,0.00008845686,0.00008018054],"domain_scores_gemma":[0.9997534,0.000044074324,0.000032256627,0.000040006624,0.000099249635,0.000030884577],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00035327827,0.000746318,0.0009578855,0.000764528,0.00052609283,0.00073991454,0.00093087903,0.0006897453,0.0012439886],"category_scores_gemma":[0.00089872896,0.0003464988,0.00057780993,0.0010383263,0.0003334262,0.0014264407,0.001610901,0.00080113835,0.00064259244],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009233421,0.0004139239,0.0070999046,0.00017289721,0.0002277888,0.00042931488,0.0003775403,0.09856034,0.10768065,0.004693744,0.005559495,0.77386105],"study_design_scores_gemma":[0.000016785729,0.00011177522,0.0040653185,0.000012964637,0.00006929362,0.00017345036,0.00009509953,0.97291845,0.017300457,0.0038552189,0.0013345083,0.000046744743],"about_ca_topic_score_codex":0.0073572956,"about_ca_topic_score_gemma":0.010444962,"teacher_disagreement_score":0.0073572956,"about_ca_system_score_codex":0.0002849792,"about_ca_system_score_gemma":0.0009777658,"threshold_uncertainty_score":0.014628947},"labels":[],"label_agreement":null},{"id":"W4403585977","doi":"10.48550/arxiv.2409.03868","title":"Few-shot Adaptation of Medical Vision-Language Models","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Adaptation (eye); Shot (pellet); Computer science; Computer vision; Artificial intelligence; Optometry; Psychology; Medicine; Neuroscience; Chemistry","score_opus":0.08143299940705799,"score_gpt":0.252326790397519,"score_spread":0.17089379099046098,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403585977","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08731147,0.0018533571,0.8948146,0.0007737765,0.00041600934,0.0002654995,0.0008800549,0.010179116,0.0035060958],"genre_scores_gemma":[0.63763386,0.00078531815,0.34443414,0.0014521289,0.00026053056,0.00035565335,0.0050236667,0.0010536134,0.009001064],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9991732,0.00026271032,0.000030245948,0.00032698162,0.00012677965,0.00008012304],"domain_scores_gemma":[0.998628,0.0007595407,0.000070371076,0.00028118328,0.00015886917,0.00010194473],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019716497,0.0013552557,0.00084448187,0.00065384485,0.0003570586,0.0009798673,0.0020215942,0.001976153,0.0026920573],"category_scores_gemma":[0.0073468834,0.00046302666,0.0011847257,0.0005703779,0.00081066735,0.0015498106,0.0020720107,0.0024987152,0.0018211302],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00083899254,0.0005153924,0.0020010872,0.00056658284,0.00035207768,0.00030758587,0.00027411757,0.40662885,0.041307393,0.0053468463,0.015337467,0.5265237],"study_design_scores_gemma":[0.000029091878,0.00014591178,0.000552371,0.000024277404,0.000028697556,0.000156496,0.00004019947,0.98097044,0.011050392,0.0051873927,0.0017876325,0.000027007341],"about_ca_topic_score_codex":0.0036292057,"about_ca_topic_score_gemma":0.0045958855,"teacher_disagreement_score":0.0036292057,"about_ca_system_score_codex":0.00073294464,"about_ca_system_score_gemma":0.00093056035,"threshold_uncertainty_score":0.010427177},"labels":[],"label_agreement":null},{"id":"W4403780086","doi":"10.48550/arxiv.2409.13675","title":"OLiVia-Nav: An Online Lifelong Vision Language Approach for Mobile Robot Social Navigation","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Computer science; Mobile robot; Mobile robot navigation; Human–computer interaction; Computer vision; Artificial intelligence; Robot; Robot control","score_opus":0.07101110598157714,"score_gpt":0.2646310933133353,"score_spread":0.19361998733175814,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403780086","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016745266,0.00025368808,0.97565496,0.00021508061,0.00008776794,0.00006184676,0.00012544464,0.0046160836,0.0022398036],"genre_scores_gemma":[0.47692272,0.00024246829,0.5100395,0.0006653148,0.00007916961,0.00026530275,0.0008539708,0.00066688575,0.010264652],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99966085,0.00008085237,0.000013635824,0.00012970765,0.000069035385,0.000046025914],"domain_scores_gemma":[0.9995838,0.00013204599,0.0000432522,0.00009408568,0.000101398546,0.000045442266],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005646846,0.00078304025,0.0005291543,0.0003428753,0.00038834458,0.00062489184,0.002197552,0.0010315728,0.0026280568],"category_scores_gemma":[0.0019092325,0.00034018082,0.00068068405,0.00024770477,0.0006165422,0.0019495831,0.0020899654,0.0016291148,0.0011021795],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002748416,0.0003789896,0.0019249296,0.00020066806,0.00011202078,0.00024160725,0.0005172985,0.27055883,0.030055715,0.02224282,0.011305851,0.66218644],"study_design_scores_gemma":[0.000013113401,0.00008959887,0.00015132874,0.000009101083,0.000010842844,0.000048241847,0.000042972057,0.98364973,0.0044507454,0.008081244,0.0034366958,0.000016360202],"about_ca_topic_score_codex":0.0061936183,"about_ca_topic_score_gemma":0.011338278,"teacher_disagreement_score":0.0061936183,"about_ca_system_score_codex":0.00066853064,"about_ca_system_score_gemma":0.0011344532,"threshold_uncertainty_score":0.0123150945},"labels":[],"label_agreement":null},{"id":"W4403842289","doi":"10.1007/978-3-031-73347-5_17","title":"Reason2Drive: Towards Interpretable and Chain-Based Reasoning for Autonomous Driving","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada)","funders":"","keywords":"Computer science; Artificial intelligence; Chain (unit)","score_opus":0.00964357850915667,"score_gpt":0.26815169255353205,"score_spread":0.2585081140443754,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403842289","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0020745653,0.0002324316,0.9771185,0.00014387879,0.000075709184,0.00008810841,0.000737594,0.016054392,0.0034748395],"genre_scores_gemma":[0.06188791,0.00047232787,0.9253162,0.00020636444,0.000052125677,0.00015508753,0.003084787,0.0021291801,0.006695933],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990029,0.000155025,0.00008239151,0.00023473374,0.00045523662,0.00006967063],"domain_scores_gemma":[0.9989423,0.00050454,0.00005008106,0.0002872505,0.00017502427,0.000040717172],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011272045,0.0017091403,0.0010370121,0.001139121,0.000736123,0.00316639,0.004492673,0.0021015343,0.022400945],"category_scores_gemma":[0.0035822287,0.0013389718,0.0031390316,0.0007784193,0.0013640397,0.004400188,0.003771405,0.0030515334,0.00595734],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004816608,0.00030546577,0.0012085594,0.0015662877,0.00034862373,0.0004167775,0.0009916185,0.21037433,0.012490839,0.17270228,0.055334393,0.54377925],"study_design_scores_gemma":[0.00007856118,0.000041811123,0.0001712865,0.00013945563,0.00009278432,0.00011526566,0.000097427794,0.73639363,0.010376587,0.20140758,0.051041856,0.00004385204],"about_ca_topic_score_codex":0.008642176,"about_ca_topic_score_gemma":0.013464165,"teacher_disagreement_score":0.022400945,"about_ca_system_score_codex":0.0007459152,"about_ca_system_score_gemma":0.0014044457,"threshold_uncertainty_score":0.074938655},"labels":[],"label_agreement":null},{"id":"W4403990800","doi":"10.1007/978-3-031-72986-7_25","title":"TIBET: Identifying and Evaluating Biases in Text-to-Image Generative Models","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of British Columbia","funders":"","keywords":"Computer science; Generative grammar; Image (mathematics); Artificial intelligence; Generative model; Natural language processing; Information retrieval","score_opus":0.0669570626640727,"score_gpt":0.353602100343215,"score_spread":0.2866450376791423,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403990800","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12512922,0.0026467822,0.8421901,0.00096079306,0.00041388604,0.0003661617,0.0030028035,0.019099602,0.00619074],"genre_scores_gemma":[0.63341546,0.00073165284,0.34505346,0.00078134757,0.00027362184,0.0005308156,0.009638886,0.0036599082,0.005914828],"study_design_codex":"simulation_or_modeling","study_design_gemma":"observational","domain_scores_codex":[0.9965848,0.0021099758,0.00013659004,0.0004849465,0.0005227335,0.00016092752],"domain_scores_gemma":[0.96485627,0.031085199,0.00056191126,0.0019680322,0.0011615627,0.00036708088],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009232542,0.001692334,0.0015775529,0.0018938619,0.00092288514,0.0029064382,0.0028991783,0.0027947568,0.005619152],"category_scores_gemma":[0.042192973,0.0009311536,0.0018080188,0.001774871,0.0013957289,0.003929481,0.0033614298,0.0039088926,0.0020756302],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00204642,0.00043159636,0.010704232,0.0006910111,0.00076733547,0.00023455912,0.00047313972,0.49469966,0.00428992,0.036004942,0.032462023,0.41719523],"study_design_scores_gemma":[0.000059830803,0.000052933152,0.00034870423,0.000028348884,0.000039027702,0.00003157246,0.000026777723,0.9706454,0.0013445201,0.026496587,0.00091385737,0.000012376163],"about_ca_topic_score_codex":0.008169123,"about_ca_topic_score_gemma":0.011053693,"teacher_disagreement_score":0.009232542,"about_ca_system_score_codex":0.0018742309,"about_ca_system_score_gemma":0.001540997,"threshold_uncertainty_score":0.048826873},"labels":[],"label_agreement":null},{"id":"W4404545360","doi":"10.1007/978-3-031-73021-4_23","title":"UniIR: Training and Benchmarking Universal Multimodal Information Retrievers","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Benchmarking; Computer science; Training (meteorology); Artificial intelligence; Human–computer interaction; Geography","score_opus":0.011900803562801286,"score_gpt":0.2410022629830654,"score_spread":0.2291014594202641,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404545360","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43406305,0.015322413,0.36657917,0.00078026863,0.0017606791,0.0015709599,0.013276558,0.14550894,0.021137826],"genre_scores_gemma":[0.49370286,0.0017345422,0.41965795,0.0008500927,0.000358782,0.0015706617,0.047739882,0.0051200013,0.029265266],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9955918,0.0013503912,0.00036056546,0.0015094277,0.00074934104,0.00043839298],"domain_scores_gemma":[0.9953511,0.0022967793,0.00016932216,0.00107401,0.0008527973,0.0002559899],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006528571,0.0028814084,0.0018566207,0.0031379843,0.0013674631,0.0021916546,0.0041543082,0.002933038,0.0122441],"category_scores_gemma":[0.012086986,0.0010823028,0.0014066986,0.0022218546,0.0009597282,0.0049869665,0.003682235,0.0021557678,0.009733526],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018372694,0.0012837674,0.004070815,0.0010084192,0.00054643484,0.00020839104,0.00031373382,0.031752218,0.02002373,0.0017221463,0.0667436,0.87048954],"study_design_scores_gemma":[0.000575398,0.002485386,0.007268573,0.00026918302,0.00053180323,0.00066349143,0.0008742216,0.8594416,0.08982854,0.005244032,0.032629237,0.0001885755],"about_ca_topic_score_codex":0.008926051,"about_ca_topic_score_gemma":0.012020742,"teacher_disagreement_score":0.0122441,"about_ca_system_score_codex":0.0016004461,"about_ca_system_score_gemma":0.0017346366,"threshold_uncertainty_score":0.04096061},"labels":[],"label_agreement":null},{"id":"W4404554547","doi":"10.2139/ssrn.5028149","title":"Visualrwkv-Hm: Enhancing Linear Visual-Language Models Via Hybrid Mixing","year":2024,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Mixing (physics); Computer science; Mathematics; Natural language processing; Artificial intelligence; Physics","score_opus":0.008782071585116355,"score_gpt":0.303193492823928,"score_spread":0.29441142123881164,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404554547","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008186873,0.00023897075,0.97847533,0.00016081867,0.0001324982,0.000047180605,0.0003058935,0.009845008,0.002607476],"genre_scores_gemma":[0.4101727,0.00041239147,0.55690145,0.0007360543,0.00016334961,0.00026905732,0.002141546,0.0032363422,0.025967127],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991571,0.00027596107,0.00003336953,0.0002312592,0.00022176674,0.00008054694],"domain_scores_gemma":[0.9989472,0.00050608895,0.00004366133,0.0002351237,0.0001920972,0.00007587892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010512542,0.0013915031,0.0008822456,0.00049719214,0.00033010144,0.0014599367,0.0021018442,0.0015196088,0.013611089],"category_scores_gemma":[0.004167099,0.000535202,0.001079431,0.00052336947,0.00045955903,0.0024145572,0.0030154036,0.002112227,0.007700335],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011270972,0.000506333,0.0007301162,0.00039944472,0.00026138738,0.00022142485,0.0001588779,0.3015197,0.057779636,0.018577114,0.015791545,0.6029274],"study_design_scores_gemma":[0.000023863166,0.000054282147,0.000056668123,0.000008177242,0.00001783713,0.0000353557,0.0000117270965,0.98221266,0.009689187,0.0059580645,0.0019187104,0.000013374971],"about_ca_topic_score_codex":0.0052189496,"about_ca_topic_score_gemma":0.00738057,"teacher_disagreement_score":0.013611089,"about_ca_system_score_codex":0.00042575347,"about_ca_system_score_gemma":0.00079485384,"threshold_uncertainty_score":0.045533597},"labels":[],"label_agreement":null},{"id":"W4404782669","doi":"10.18653/v1/2024.emnlp-main.944","title":"On Efficient Language and Vision Assistants for Visually-Situated Natural Language Understanding: What Matters in Reading and Reasoning","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Korea Advanced Institute of Science and Technology","keywords":"Situated; Computer science; Reading (process); Natural language; Natural (archaeology); Natural language processing; Artificial intelligence; Human–computer interaction; Linguistics","score_opus":0.012004506382721153,"score_gpt":0.3344200225874453,"score_spread":0.32241551620472414,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404782669","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040659707,0.0007153068,0.938402,0.004372981,0.00015886818,0.00019795557,0.0008389008,0.009649334,0.0050049718],"genre_scores_gemma":[0.30230266,0.0007520342,0.68438673,0.0009928223,0.00013188577,0.0003330423,0.002575985,0.0018249317,0.0066999253],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984459,0.00057455816,0.00008774842,0.00049563707,0.00022806771,0.0001681074],"domain_scores_gemma":[0.9916957,0.0046502645,0.00030272926,0.0021196299,0.000865289,0.0003664135],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030301907,0.00094924023,0.00104536,0.00061873376,0.0011552426,0.003550209,0.004230631,0.0021093616,0.011625579],"category_scores_gemma":[0.021395283,0.00097414263,0.0016246933,0.00083700265,0.001787092,0.013682951,0.0034951712,0.0040766103,0.0058429376],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008083606,0.0008913978,0.004291163,0.0007985734,0.0001483806,0.00023952565,0.0010625157,0.2343776,0.010711428,0.1181833,0.046541244,0.58194643],"study_design_scores_gemma":[0.00006136251,0.00007792932,0.00041471494,0.0000728693,0.00003735376,0.00009129395,0.00022035619,0.84354323,0.0056918664,0.1410705,0.0086886585,0.000029950535],"about_ca_topic_score_codex":0.009944311,"about_ca_topic_score_gemma":0.019553047,"teacher_disagreement_score":0.011625579,"about_ca_system_score_codex":0.0019075396,"about_ca_system_score_gemma":0.0039455695,"threshold_uncertainty_score":0.038891494},"labels":[],"label_agreement":null},{"id":"W4404783412","doi":"10.18653/v1/2022.aacl-main.61","title":"A Prompt Array Keeps the Bias Away: Debiasing Vision-Language Models with Adversarial Learning","year":2022,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Economic and Social Research Council; Engineering and Physical Sciences Research Council","keywords":"Debiasing; Adversarial system; Computer science; Artificial intelligence; Computer vision; Natural language processing; Psychology; Cognitive science","score_opus":0.017603201543926292,"score_gpt":0.26099538542813217,"score_spread":0.24339218388420586,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404783412","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034043815,0.0014943541,0.95635253,0.002213655,0.0003879356,0.00005720834,0.00007851937,0.0015724734,0.003799484],"genre_scores_gemma":[0.82885396,0.00059679354,0.1578541,0.00205657,0.00032212745,0.00016823687,0.00019572744,0.0005982018,0.009354305],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99932075,0.00030976068,0.000023822617,0.00014798692,0.000107752145,0.00008991111],"domain_scores_gemma":[0.9966318,0.0023242529,0.0001971835,0.00041499673,0.0002616221,0.00017007101],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028581794,0.0011639505,0.0011833389,0.00034780556,0.00064720423,0.0009884749,0.0021695266,0.001866059,0.0036108133],"category_scores_gemma":[0.010404047,0.00083228166,0.000572477,0.00033945093,0.0017660712,0.002831743,0.0037375642,0.003933096,0.0010168513],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007096048,0.00014357745,0.0010848958,0.0001266344,0.00014127197,0.0001649414,0.00032449866,0.76963997,0.0077775437,0.034638684,0.014579658,0.17066875],"study_design_scores_gemma":[0.000026424827,0.000043006126,0.000045547844,0.000011325069,0.000010403355,0.000019380406,0.000011407667,0.9829137,0.0012043798,0.015147833,0.0005579186,0.000008611829],"about_ca_topic_score_codex":0.002973559,"about_ca_topic_score_gemma":0.0033916382,"teacher_disagreement_score":0.0036108133,"about_ca_system_score_codex":0.00080066506,"about_ca_system_score_gemma":0.0009980857,"threshold_uncertainty_score":0.015115678},"labels":[],"label_agreement":null},{"id":"W4404783635","doi":"10.18653/v1/2024.findings-emnlp.83","title":"EchoSight: Advancing Visual-Language Models with Wiki Knowledge","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Visual language; Human–computer interaction; Natural language processing; Knowledge management; World Wide Web; Linguistics","score_opus":0.007573520661421284,"score_gpt":0.3036279570150422,"score_spread":0.2960544363536209,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404783635","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016683685,0.0021530804,0.94783413,0.0007805675,0.00023415574,0.00033851428,0.002111967,0.025175972,0.004687894],"genre_scores_gemma":[0.2957205,0.0013060769,0.6807813,0.0014289819,0.00025380595,0.00058796164,0.011398901,0.0012033619,0.007319035],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979387,0.00082351995,0.000120389166,0.0006133933,0.00039730794,0.00010666387],"domain_scores_gemma":[0.99325037,0.0044451216,0.00026315887,0.0010758431,0.0007492574,0.00021617666],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030685866,0.0016527048,0.0010103933,0.002422878,0.00054277235,0.00244464,0.0033151994,0.0017625073,0.004420634],"category_scores_gemma":[0.013464525,0.00056202087,0.0017589914,0.0010388483,0.0009740963,0.0058621415,0.003338928,0.0026080867,0.0030701316],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045052744,0.00073259126,0.0033638126,0.0011881364,0.00035578766,0.0003535229,0.0007968768,0.12007434,0.011886725,0.02533487,0.042457953,0.7930049],"study_design_scores_gemma":[0.00009469484,0.00015690559,0.00038721605,0.00008673947,0.00007324334,0.00019005996,0.00018496814,0.93498135,0.005647495,0.04230615,0.01583632,0.000054787506],"about_ca_topic_score_codex":0.012068933,"about_ca_topic_score_gemma":0.015977157,"teacher_disagreement_score":0.012068933,"about_ca_system_score_codex":0.0010279068,"about_ca_system_score_gemma":0.0017976057,"threshold_uncertainty_score":0.023997366},"labels":[],"label_agreement":null},{"id":"W4404792981","doi":"10.18653/v1/2024.emnlp-main.633","title":"TroL: Traversal of Layers for Large Language and Vision Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"National Supercomputing Center, Korea Institute of Science and Technology Information; Institute for Information and Communications Technology Promotion; Defense Acquisition Program Administration; Ministry of Science and ICT, South Korea; Korea Institute of Science and Technology Information","keywords":"Tree traversal; Computer science; Artificial intelligence; Programming language","score_opus":0.010395003013021922,"score_gpt":0.32150961418095675,"score_spread":0.3111146111679348,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404792981","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026664361,0.0012112478,0.85324115,0.0010098565,0.0004212666,0.00024998776,0.0021595934,0.10782843,0.007214087],"genre_scores_gemma":[0.3067977,0.0005469388,0.6577886,0.0013729488,0.00010360848,0.00069172843,0.00723539,0.012994672,0.012468408],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9993062,0.00013598573,0.000060457412,0.00022776864,0.00013175538,0.00013772481],"domain_scores_gemma":[0.9987251,0.0005550896,0.00007864998,0.00041407213,0.00014268263,0.00008441268],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009525052,0.0018989902,0.0008127538,0.0007752964,0.0006770271,0.0019711722,0.0031928532,0.0016185328,0.012313846],"category_scores_gemma":[0.0055661835,0.0010458895,0.00211221,0.000584336,0.0009413793,0.004572616,0.0028015263,0.0036028754,0.0043122284],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00069367426,0.00025826602,0.003889889,0.000946319,0.00038022394,0.0005200352,0.0005383506,0.30835587,0.028813116,0.05945267,0.120364316,0.47578725],"study_design_scores_gemma":[0.000059720463,0.0000513015,0.00016990448,0.000045479053,0.000035824127,0.000087194414,0.000042721418,0.94102347,0.007827142,0.036539305,0.014089326,0.000028555982],"about_ca_topic_score_codex":0.0122471405,"about_ca_topic_score_gemma":0.024303224,"teacher_disagreement_score":0.012313846,"about_ca_system_score_codex":0.002212505,"about_ca_system_score_gemma":0.002897566,"threshold_uncertainty_score":0.041193902},"labels":[],"label_agreement":null},{"id":"W4404820294","doi":"10.1007/978-3-031-72848-8_26","title":"CIC-BART-SSA: Controllable Image Captioning with Structured Semantic Augmentation","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Centre for Social Innovation","funders":"","keywords":"Closed captioning; Computer science; Image (mathematics); Natural language processing; Artificial intelligence; Information retrieval","score_opus":0.008056717923322488,"score_gpt":0.2529892578538221,"score_spread":0.2449325399304996,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404820294","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004112376,0.0005060121,0.93304724,0.00010597207,0.00043708875,0.00023172003,0.0017227231,0.047902305,0.011934589],"genre_scores_gemma":[0.07848524,0.00045199203,0.8942213,0.00030093268,0.00023872517,0.000441712,0.0077365227,0.0057578273,0.012365692],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99935,0.000101145866,0.000034773373,0.00019733231,0.00025415752,0.000062595704],"domain_scores_gemma":[0.9992223,0.0002241848,0.000032312226,0.00030615233,0.00016242023,0.000052642095],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005721565,0.002134628,0.0012365845,0.0010272416,0.0005049194,0.001672399,0.0022659579,0.0014071628,0.02914339],"category_scores_gemma":[0.0017063876,0.0007826903,0.00145964,0.0012299041,0.00079169864,0.0020994751,0.002875094,0.0019390205,0.015900802],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007418647,0.00023295727,0.00015758062,0.00079417555,0.00010471147,0.00041791343,0.00016353898,0.018332992,0.13841946,0.021223992,0.10805663,0.71135426],"study_design_scores_gemma":[0.00014737608,0.00029612475,0.0004843298,0.00010638127,0.00008722439,0.0007441643,0.000087845314,0.6878115,0.16178185,0.048069526,0.10025893,0.0001246504],"about_ca_topic_score_codex":0.0018961473,"about_ca_topic_score_gemma":0.0028085609,"teacher_disagreement_score":0.02914339,"about_ca_system_score_codex":0.00038925785,"about_ca_system_score_gemma":0.0006882397,"threshold_uncertainty_score":0.097494364},"labels":[],"label_agreement":null},{"id":"W4404893151","doi":"10.1007/978-3-031-72897-6_19","title":"VITATECS: A Diagnostic Dataset for Temporal Concept Understanding of Video-Language Models","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada)","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence","score_opus":0.03741656571281926,"score_gpt":0.3051162996673086,"score_spread":0.2676997339544893,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404893151","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0646027,0.0023719585,0.043967616,0.00084336865,0.0006596195,0.0014053785,0.8329862,0.045019228,0.008143941],"genre_scores_gemma":[0.03810112,0.00040310982,0.036447987,0.00022269228,0.00005353072,0.00070840004,0.9197385,0.00076472363,0.003559928],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9990029,0.0001909557,0.000094852316,0.0003288474,0.00027692653,0.00010567885],"domain_scores_gemma":[0.9974146,0.0011087239,0.00014898682,0.0006234108,0.00045906557,0.00024524753],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012543094,0.0029898495,0.0010160839,0.003380964,0.0008282922,0.0016133876,0.0034305267,0.0031212545,0.01483277],"category_scores_gemma":[0.007533285,0.0006145686,0.0018594743,0.0024865652,0.00063222967,0.0018969056,0.0017090889,0.0026931828,0.012200166],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014659916,0.0012187057,0.0058359625,0.0022613297,0.0004642171,0.0009404557,0.00028439605,0.017987473,0.014179532,0.0040466725,0.7580483,0.19326705],"study_design_scores_gemma":[0.0019795627,0.0013295854,0.024643097,0.0007771599,0.00063811085,0.0024955533,0.0013356023,0.364022,0.0537208,0.019772636,0.528804,0.00048196543],"about_ca_topic_score_codex":0.030348476,"about_ca_topic_score_gemma":0.043138083,"teacher_disagreement_score":0.030348476,"about_ca_system_score_codex":0.0018368941,"about_ca_system_score_gemma":0.0024537423,"threshold_uncertainty_score":0.060343683},"labels":[],"label_agreement":null},{"id":"W4404999448","doi":"10.3390/info15120766","title":"Enabling Perspective-Aware Ai with Contextual Scene Graph Generation","year":2024,"lang":"en","type":"article","venue":"Information","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Contextual design; Artificial intelligence; Graph; Human–computer interaction; Perspective (graphical); Data science; Natural language processing; Object (grammar); Theoretical computer science","score_opus":0.012782248264085423,"score_gpt":0.2756536029919525,"score_spread":0.2628713547278671,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404999448","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010283341,0.00030454976,0.9697028,0.00027434228,0.00006923902,0.000117111325,0.0007704973,0.015334703,0.0031434803],"genre_scores_gemma":[0.2199177,0.0003716232,0.76840377,0.00052578404,0.000054133285,0.00020025068,0.0046688626,0.0024973743,0.003360398],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995653,0.00010297989,0.000015225414,0.00016648594,0.00010730827,0.000042628304],"domain_scores_gemma":[0.9994498,0.00025590177,0.000025239624,0.00014027009,0.00008780079,0.000040959163],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003390475,0.0014614057,0.00062653143,0.00084944034,0.0003932594,0.00090320973,0.0017735005,0.00097099406,0.006102355],"category_scores_gemma":[0.002110917,0.00046125773,0.0016575325,0.00059454254,0.00055666454,0.0017737309,0.0020394118,0.001623593,0.0021154569],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042019115,0.0002491645,0.0016722339,0.00063843705,0.0001880657,0.0008326425,0.0008862926,0.29153255,0.06316943,0.02907657,0.030686432,0.580648],"study_design_scores_gemma":[0.000027824319,0.000043529682,0.00027310138,0.000022471488,0.000024799012,0.00014185227,0.00014669246,0.94909626,0.012254826,0.026538817,0.011404662,0.000025258314],"about_ca_topic_score_codex":0.00517215,"about_ca_topic_score_gemma":0.010925434,"teacher_disagreement_score":0.006102355,"about_ca_system_score_codex":0.0005818351,"about_ca_system_score_gemma":0.0006041379,"threshold_uncertainty_score":0.020414412},"labels":[],"label_agreement":null},{"id":"W4405023785","doi":"10.1109/cvpr52734.2025.01822","title":"All Languages Matter: Evaluating LMMs on Culturally Diverse 100 Languages","year":2025,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Benchmark (surveying); Computer science; Resource (disambiguation); Artificial intelligence; Data science; Natural language processing; Geography","score_opus":0.022328242076495475,"score_gpt":0.3794272945488076,"score_spread":0.3570990524723121,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405023785","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.748957,0.005668902,0.11899618,0.004846936,0.0011637044,0.0032968451,0.030832691,0.013923989,0.07231385],"genre_scores_gemma":[0.79917747,0.0009772932,0.118176796,0.002652762,0.00017509723,0.0034842638,0.0647361,0.001201069,0.009419125],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99197376,0.0055589117,0.00041159886,0.0009990414,0.0008070339,0.0002496353],"domain_scores_gemma":[0.9796007,0.01563732,0.0004731032,0.0018968215,0.0016975364,0.0006944206],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009059776,0.0027386867,0.00089915894,0.0016542044,0.0012431336,0.0031734505,0.0029985257,0.0026975076,0.013269286],"category_scores_gemma":[0.03964505,0.00046445229,0.0020916327,0.0013590312,0.0013092085,0.0064506563,0.0048220204,0.0029741093,0.004732336],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0034653528,0.0042970246,0.054572016,0.005314319,0.0014261438,0.0006875534,0.005699627,0.2129634,0.007737758,0.024613177,0.14904429,0.5301793],"study_design_scores_gemma":[0.00071900635,0.002259319,0.023468692,0.0015658474,0.0005554869,0.0005114658,0.008331732,0.8339854,0.0117971655,0.05305268,0.063388266,0.00036495793],"about_ca_topic_score_codex":0.0148893865,"about_ca_topic_score_gemma":0.019729892,"teacher_disagreement_score":0.0148893865,"about_ca_system_score_codex":0.002836028,"about_ca_system_score_gemma":0.001988583,"threshold_uncertainty_score":0.047913253},"labels":[],"label_agreement":null},{"id":"W4405140008","doi":"10.1007/978-981-96-0917-8_17","title":"Vision Language Models are blind","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing; Computer vision; Speech recognition","score_opus":0.017041941641969042,"score_gpt":0.2930850625544102,"score_spread":0.27604312091244115,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405140008","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012343852,0.009024541,0.8123169,0.014235164,0.0021716792,0.000034125576,0.0010108853,0.0060891407,0.14277376],"genre_scores_gemma":[0.5426603,0.011753228,0.14112149,0.0069263247,0.0023309328,0.00011902958,0.0025645364,0.0028116496,0.28971252],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99915147,0.00019176524,0.000038404803,0.00025366782,0.00027382973,0.00009089545],"domain_scores_gemma":[0.99764097,0.0011775715,0.00013221383,0.0005835134,0.0003866042,0.00007913913],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010535341,0.00089283165,0.0009846973,0.0006185666,0.0006196969,0.003626584,0.001090565,0.0021545957,0.011768618],"category_scores_gemma":[0.007967066,0.00090472214,0.0007072523,0.0006039462,0.0022127167,0.007022109,0.0017877708,0.0041784407,0.0088717695],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018816575,0.00006083706,0.0004346289,0.00030642375,0.000091968686,0.00018973011,0.00030036876,0.019500002,0.0073454683,0.554504,0.1023503,0.31472814],"study_design_scores_gemma":[0.0000148881845,0.000020324442,0.000237909,0.00005420249,0.000029058661,0.00026073083,0.00007634924,0.07755405,0.005437205,0.8718761,0.0444014,0.000037651396],"about_ca_topic_score_codex":0.003286597,"about_ca_topic_score_gemma":0.0019255274,"teacher_disagreement_score":0.011768618,"about_ca_system_score_codex":0.0008019052,"about_ca_system_score_gemma":0.0009963608,"threshold_uncertainty_score":0.03936994},"labels":[],"label_agreement":null},{"id":"W4405414691","doi":"10.1007/s00371-024-03729-0","title":"Dynamic text prompt joint multimodal features for accurate plant disease image captioning","year":2024,"lang":"en","type":"article","venue":"The Visual Computer","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Science Foundation of Hebei Province; National Natural Science Foundation of China","keywords":"Closed captioning; Computer science; Feature (linguistics); Artificial intelligence; Plant disease; Image (mathematics); Machine learning; Natural language processing","score_opus":0.01490606545697368,"score_gpt":0.31847195957354424,"score_spread":0.30356589411657053,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405414691","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08264563,0.005525484,0.74905604,0.0021113236,0.004259675,0.0012335948,0.026830403,0.09961146,0.028726485],"genre_scores_gemma":[0.3682091,0.0026105146,0.56103694,0.0012146651,0.001543034,0.0010822517,0.030318782,0.0042161173,0.029768616],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99959856,0.00008127545,0.000024917541,0.000107597545,0.00011123359,0.00007648991],"domain_scores_gemma":[0.99859923,0.00036243012,0.00009468053,0.0001931851,0.00065515185,0.000095322794],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005888276,0.0018007881,0.0008140012,0.0020321936,0.00045415768,0.0013305494,0.0009944864,0.0018169302,0.03285933],"category_scores_gemma":[0.0032390547,0.00035459048,0.0007178858,0.00096937787,0.000248551,0.0015689043,0.001327356,0.001355219,0.01886473],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019270487,0.00024765238,0.0007036415,0.0009102911,0.000051759747,0.0005060548,0.000098922515,0.007976475,0.14983505,0.0013233755,0.1058119,0.73060787],"study_design_scores_gemma":[0.0001249682,0.0005543262,0.0060565737,0.00024877317,0.00016604454,0.00082245376,0.00028379945,0.69349563,0.21008588,0.0050881687,0.082945615,0.00012770927],"about_ca_topic_score_codex":0.0017526787,"about_ca_topic_score_gemma":0.0026548265,"teacher_disagreement_score":0.03285933,"about_ca_system_score_codex":0.00047367008,"about_ca_system_score_gemma":0.0005706974,"threshold_uncertainty_score":0.10992539},"labels":[],"label_agreement":null},{"id":"W4405436672","doi":"10.1016/j.neucom.2024.129177","title":"A simple yet effective knowledge guided method for entity-aware video captioning on a basketball benchmark","year":2024,"lang":"en","type":"article","venue":"Neurocomputing","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Beijing Postdoctoral Science Foundation; Natural Science Foundation of Beijing Municipality; China Postdoctoral Science Foundation; National Natural Science Foundation of China","keywords":"Closed captioning; Basketball; Computer science; Benchmark (surveying); Simple (philosophy); Artificial intelligence; Machine learning; Information retrieval; Natural language processing; Image (mathematics)","score_opus":0.019089827883115702,"score_gpt":0.36385435458996085,"score_spread":0.34476452670684515,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405436672","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05643087,0.0033660615,0.85539335,0.0009028254,0.0011954267,0.0009424845,0.0076841363,0.057200443,0.016884374],"genre_scores_gemma":[0.29296947,0.00078432285,0.66724956,0.0006117792,0.00044639278,0.00046506434,0.02061403,0.0013955038,0.015463892],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99890983,0.00017317911,0.000063194195,0.00045248284,0.00024854287,0.00015272075],"domain_scores_gemma":[0.99895525,0.00030594534,0.00006288882,0.00016050314,0.00042329048,0.000092109534],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00078244274,0.0026640098,0.001424971,0.0026709782,0.0009733444,0.0017487803,0.0025000337,0.0022378871,0.01233402],"category_scores_gemma":[0.0032462534,0.00044764386,0.0010217207,0.0020157297,0.00040174558,0.0017713206,0.0017034911,0.0019033997,0.00597866],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00053037953,0.0003604526,0.00081102765,0.0002964646,0.000108859145,0.0002193948,0.00007696166,0.022414334,0.02426471,0.0015423977,0.051597256,0.89777774],"study_design_scores_gemma":[0.000086110165,0.00019348992,0.0012763687,0.000041536638,0.000091543414,0.00019321425,0.00013680432,0.9593303,0.023379391,0.0036229643,0.011598384,0.000049960418],"about_ca_topic_score_codex":0.013651833,"about_ca_topic_score_gemma":0.020950489,"teacher_disagreement_score":0.013651833,"about_ca_system_score_codex":0.0008879573,"about_ca_system_score_gemma":0.0017976656,"threshold_uncertainty_score":0.041261375},"labels":[],"label_agreement":null},{"id":"W4405766356","doi":"10.48550/arxiv.2412.16694","title":"DragonVerseQA: Open-Domain Long-Form Context-Aware Question-Answering","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Context (archaeology); Open domain; Question answering; Domain (mathematical analysis); Computer science; Information retrieval; History; Mathematics; Archaeology","score_opus":0.054690372831736674,"score_gpt":0.2341087230419267,"score_spread":0.17941835021019004,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405766356","genre_codex":"dataset","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.071857184,0.011046967,0.22445339,0.0056038895,0.0015155913,0.0037663158,0.560334,0.0859258,0.035496835],"genre_scores_gemma":[0.092117935,0.0010305208,0.24148536,0.0018365753,0.00036412638,0.0028246448,0.6508088,0.0013335237,0.008198582],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9958567,0.001535391,0.00038087368,0.0012450077,0.0007468996,0.00023510284],"domain_scores_gemma":[0.9928565,0.003195872,0.0004689823,0.0017127775,0.0012130745,0.0005528231],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002722518,0.00184039,0.0010357946,0.0054637007,0.0015395362,0.0030480032,0.0027845718,0.0028210832,0.008375232],"category_scores_gemma":[0.01595555,0.0005384997,0.0013719179,0.0030080944,0.000862047,0.0056579686,0.005694775,0.0028933892,0.008257809],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012831936,0.0011348712,0.01286126,0.0090177795,0.0004066097,0.00092903315,0.0072827972,0.013907123,0.028452385,0.026256487,0.6182753,0.2801932],"study_design_scores_gemma":[0.00021895178,0.0003598632,0.011993541,0.0005981526,0.00011609936,0.00061844324,0.0038955302,0.07206289,0.011887764,0.023268282,0.8747893,0.00019124488],"about_ca_topic_score_codex":0.010083126,"about_ca_topic_score_gemma":0.025753494,"teacher_disagreement_score":0.010083126,"about_ca_system_score_codex":0.0015321157,"about_ca_system_score_gemma":0.0020530527,"threshold_uncertainty_score":0.028017938},"labels":[],"label_agreement":null},{"id":"W4405812114","doi":"10.1109/tpami.2024.3522295","title":"A Review of Deep Learning for Video Captioning","year":2024,"lang":"en","type":"review","venue":"IEEE Transactions on Pattern Analysis and Machine Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Closed captioning; Computer science; Artificial intelligence; Deep learning; Computer vision; Natural language processing; Multimedia; Image (mathematics)","score_opus":0.038680409321068095,"score_gpt":0.3533350145359122,"score_spread":0.3146546052148441,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405812114","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00051732094,0.9843301,0.008838804,0.0007168289,0.0004868483,0.000028747932,0.00014992271,0.00008689068,0.0048445845],"genre_scores_gemma":[0.0044139065,0.98442876,0.0070444006,0.0005028509,0.0005786122,0.000039687136,0.00033536405,0.000035809502,0.0026206765],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996276,0.0000722259,0.000046444005,0.00008564589,0.00013862406,0.000029420145],"domain_scores_gemma":[0.9986099,0.00079985213,0.00007229715,0.00004714211,0.00042519078,0.000045604807],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011523623,0.001191512,0.00091164035,0.00269186,0.00029794822,0.0011488298,0.0014553928,0.0011636169,0.0060274573],"category_scores_gemma":[0.0033308293,0.0006093941,0.0006580837,0.0037170916,0.0004460585,0.002466867,0.0008024632,0.0015725774,0.00344023],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000039062146,0.000060631457,0.00025795528,0.0094545,0.00007368459,0.00006569336,0.000051228297,0.0025191922,0.0009157933,0.007961229,0.04185257,0.93674856],"study_design_scores_gemma":[0.000016796965,0.00016123278,0.0012973698,0.006073137,0.00018975213,0.00064116815,0.00008667963,0.0065149497,0.0022196113,0.0081822965,0.97455436,0.000062673644],"about_ca_topic_score_codex":0.0032188932,"about_ca_topic_score_gemma":0.0034876561,"teacher_disagreement_score":0.0060274573,"about_ca_system_score_codex":0.0009151598,"about_ca_system_score_gemma":0.0015798376,"threshold_uncertainty_score":0.020163834},"labels":[],"label_agreement":null},{"id":"W4405882647","doi":"10.1007/978-981-96-2061-6_31","title":"MM-CARP: Multimodal Model with Cross-Modal Retrieval-Augmented and Visual Region Perception","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Computer science; Modal; Perception; Artificial intelligence; Computer vision; Pattern recognition (psychology)","score_opus":0.014400977277997834,"score_gpt":0.2924754395763145,"score_spread":0.27807446229831667,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405882647","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005575013,0.0008008875,0.98553944,0.00013418283,0.00015753167,0.000056827208,0.0005213852,0.0042631826,0.0029514874],"genre_scores_gemma":[0.31789932,0.0015276949,0.65276366,0.00043822863,0.00031477248,0.0003750948,0.0031644371,0.0014122267,0.02210454],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996984,0.000058677044,0.000012666755,0.00010976103,0.000079039506,0.0000414863],"domain_scores_gemma":[0.99976796,0.000053962645,0.000015172414,0.00007040171,0.00007274027,0.000019823145],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005957096,0.0011094238,0.0011735237,0.00053310307,0.00031052405,0.0012503904,0.0024294055,0.0014226625,0.008116187],"category_scores_gemma":[0.0011452849,0.0005023245,0.0013439704,0.00088997866,0.00041980535,0.001718411,0.0015792241,0.0013059884,0.0041248696],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007053876,0.00024644946,0.0004884584,0.00041010932,0.00028883788,0.0002745983,0.0000955751,0.303174,0.049760938,0.018576708,0.027914256,0.5980647],"study_design_scores_gemma":[0.000012763873,0.000047759124,0.00020091895,0.000012946959,0.000030020072,0.0000753038,0.000010527127,0.98654777,0.0048363847,0.00467096,0.00353499,0.000019675388],"about_ca_topic_score_codex":0.007208166,"about_ca_topic_score_gemma":0.0062258905,"teacher_disagreement_score":0.008116187,"about_ca_system_score_codex":0.0004615225,"about_ca_system_score_gemma":0.0006718036,"threshold_uncertainty_score":0.027151346},"labels":[],"label_agreement":null},{"id":"W4406343535","doi":"10.1007/s43621-025-00815-8","title":"A comprehensive review of large language models: issues and solutions in learning environments","year":2025,"lang":"en","type":"review","venue":"Discover Sustainability","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":77,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Moncton","funders":"University of Johannesburg","keywords":"Computer science; Management science; Engineering","score_opus":0.021348942387629334,"score_gpt":0.3679048077687132,"score_spread":0.34655586538108385,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406343535","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00018303741,0.99484044,0.0012007024,0.0010576062,0.0001599129,0.000012395276,0.0000577798,0.000020241843,0.0024677634],"genre_scores_gemma":[0.0020367452,0.99513006,0.001610676,0.00035019187,0.00015023045,0.000027545273,0.00008394847,0.000011607687,0.00059897284],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985886,0.0005284515,0.0002464363,0.00016739422,0.00040572634,0.000063375],"domain_scores_gemma":[0.99306244,0.0055483994,0.0003642932,0.00018715787,0.0007294864,0.00010832135],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031138062,0.00096342433,0.0016094216,0.004436827,0.0005387484,0.002734516,0.0013995209,0.0017936707,0.006617499],"category_scores_gemma":[0.008986431,0.0005397692,0.0010859992,0.005797327,0.0010334547,0.0047496744,0.001409524,0.0017056455,0.0021997069],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000023921853,0.00004330131,0.00029213182,0.039871253,0.00012140956,0.000085605876,0.0003513796,0.0010378611,0.00030643356,0.02526461,0.024551513,0.9080506],"study_design_scores_gemma":[0.0000050420695,0.000046172034,0.0006190908,0.022677224,0.00016601157,0.00029719688,0.0002502472,0.00033273397,0.00024014158,0.010311781,0.96502656,0.000027849803],"about_ca_topic_score_codex":0.0033987146,"about_ca_topic_score_gemma":0.005497585,"teacher_disagreement_score":0.006617499,"about_ca_system_score_codex":0.0014465317,"about_ca_system_score_gemma":0.0056877485,"threshold_uncertainty_score":0.022137702},"labels":[],"label_agreement":null},{"id":"W4406365662","doi":"10.2139/ssrn.5097096","title":"Lifelong Scene Graph Generation","year":2025,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Graph; Computer science; Theoretical computer science","score_opus":0.013710368100805935,"score_gpt":0.28395544944492856,"score_spread":0.2702450813441226,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406365662","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036959372,0.0005363402,0.88428324,0.0008428213,0.0005961518,0.0006467462,0.008317877,0.03698491,0.030832492],"genre_scores_gemma":[0.30828372,0.00037829042,0.6128838,0.0006771829,0.00017650233,0.00050793734,0.0301297,0.007713762,0.039249033],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996505,0.000050686584,0.000011070433,0.00012871998,0.00010611074,0.000052932886],"domain_scores_gemma":[0.9993754,0.00012482925,0.000016600954,0.0002840969,0.00015000481,0.000049090653],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002849534,0.0009960044,0.00062668405,0.0012212371,0.00062480604,0.0010732695,0.0019400186,0.0013903838,0.044107918],"category_scores_gemma":[0.0015948951,0.0006200422,0.0011692363,0.0009327813,0.00040120631,0.0013350637,0.0019396722,0.0013440287,0.011466765],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004932085,0.00036664715,0.0014586755,0.0004855864,0.00014975418,0.0005338415,0.00020739542,0.06789653,0.033490717,0.03234527,0.17600864,0.6865637],"study_design_scores_gemma":[0.00014027732,0.00016125353,0.0007888715,0.000045723413,0.00007586643,0.00042861863,0.00014677877,0.8187051,0.03159553,0.07001109,0.0778596,0.000041239527],"about_ca_topic_score_codex":0.0030750497,"about_ca_topic_score_gemma":0.009251508,"teacher_disagreement_score":0.044107918,"about_ca_system_score_codex":0.0006264691,"about_ca_system_score_gemma":0.00073253474,"threshold_uncertainty_score":0.14755565},"labels":[],"label_agreement":null},{"id":"W4406612097","doi":"10.1109/smc54092.2024.10831147","title":"TransUAAE-CapGen: Caption Generation from Histopathological Patches through Transformer and UNet-Based Adversarial Autoencoder","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Autoencoder; Adversarial system; Computer science; Transformer; Artificial intelligence; Computer vision; Deep learning; Engineering; Electrical engineering","score_opus":0.03054306869164053,"score_gpt":0.27079084329428615,"score_spread":0.24024777460264563,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406612097","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017084451,0.0008208595,0.9681065,0.00031420775,0.0003574587,0.0002052924,0.00050276594,0.010223139,0.0023854398],"genre_scores_gemma":[0.34044164,0.001068822,0.6395466,0.00094587327,0.00026173433,0.0004069582,0.0040076543,0.001264466,0.012056362],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9995384,0.000088656925,0.000024340165,0.00016856538,0.0001307562,0.00004940623],"domain_scores_gemma":[0.9991658,0.0003257445,0.000067628076,0.00018847379,0.00021202746,0.000040409304],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001055188,0.0014391423,0.0006055324,0.00076597417,0.00024379844,0.00077594724,0.0012650458,0.0012618846,0.0036888584],"category_scores_gemma":[0.0029238167,0.00041803427,0.0010864364,0.0004153047,0.0005853544,0.0011904463,0.0014323954,0.0016738,0.0019626776],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041228958,0.00014005684,0.0011607567,0.0003717869,0.00018243771,0.00070989307,0.00018372276,0.19858429,0.077396624,0.0042120023,0.021848846,0.6947973],"study_design_scores_gemma":[0.000018252094,0.000117717515,0.00045620213,0.000027279471,0.000039357325,0.0003897243,0.000027789589,0.951012,0.039811492,0.0024635291,0.005612131,0.000024489216],"about_ca_topic_score_codex":0.0013967421,"about_ca_topic_score_gemma":0.002029876,"teacher_disagreement_score":0.0036888584,"about_ca_system_score_codex":0.0004916483,"about_ca_system_score_gemma":0.00055974134,"threshold_uncertainty_score":0.012340486},"labels":[],"label_agreement":null},{"id":"W4406800169","doi":"10.1007/978-3-031-78554-2_17","title":"Medical Report Generation from Medical Images Using Vision Transformer and Bart Deep Learning Architectures","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Artificial intelligence; Transformer; Computer vision; Deep learning; Medical imaging; Electrical engineering; Engineering","score_opus":0.011354863637457269,"score_gpt":0.2946669953497657,"score_spread":0.28331213171230846,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406800169","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012063045,0.000472075,0.9690606,0.0005384652,0.00037327965,0.00025403636,0.0015870598,0.011918613,0.0037329767],"genre_scores_gemma":[0.24651164,0.0008691879,0.73273045,0.00038741936,0.00034237193,0.00031783583,0.0057679424,0.0012741927,0.011798975],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99965346,0.000044890156,0.00002581685,0.000079852754,0.00016111559,0.00003485732],"domain_scores_gemma":[0.99914443,0.000312236,0.000059648883,0.0001539942,0.00027918382,0.000050492203],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008123492,0.0009099863,0.0006099746,0.0016392393,0.00019141925,0.0009886901,0.000967425,0.00078604673,0.008751392],"category_scores_gemma":[0.003040294,0.00039202292,0.0008901211,0.0008028161,0.00024145209,0.0007087249,0.0009471523,0.0009806228,0.004374963],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005986842,0.00012033967,0.0009251059,0.00019997795,0.00004873545,0.00045817293,0.00005575953,0.033660475,0.03254642,0.0046877903,0.028570697,0.8981279],"study_design_scores_gemma":[0.00006283135,0.0001406106,0.00084988045,0.000041940653,0.00005197068,0.00069810025,0.000041332973,0.9038515,0.0697587,0.010624884,0.0138380155,0.000040245366],"about_ca_topic_score_codex":0.0021157528,"about_ca_topic_score_gemma":0.0025633543,"teacher_disagreement_score":0.008751392,"about_ca_system_score_codex":0.00054181484,"about_ca_system_score_gemma":0.0007774726,"threshold_uncertainty_score":0.029276311},"labels":[],"label_agreement":null},{"id":"W4407046109","doi":"10.1111/area.12996","title":"Visualising an undergraduate geography field class using generative AI: Intent, expectations and surprises about the racial depiction of students","year":2025,"lang":"en","type":"article","venue":"Area","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Okanagan College; Simon Fraser University","funders":"","keywords":"Surprise; Depiction; Generative grammar; Narrative; Field (mathematics); Class (philosophy); Sociology; Epistemology; Intentionality; Computer science; Visual arts; Artificial intelligence; Philosophy; Communication; Art","score_opus":0.024255905152098346,"score_gpt":0.371160724422869,"score_spread":0.34690481927077066,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407046109","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.79168963,0.0004968468,0.035840396,0.029857205,0.0015573344,0.00015473292,0.00021981435,0.0009022484,0.13928182],"genre_scores_gemma":[0.97866184,0.00013790026,0.002680546,0.0013156013,0.00015065659,0.000052437666,0.000036713078,0.00017008041,0.016794212],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.99812394,0.0014033158,0.000024580362,0.00011573249,0.00017596647,0.00015638786],"domain_scores_gemma":[0.99198157,0.0066073085,0.0002915523,0.00026153878,0.00032102456,0.0005369548],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025894677,0.00059221894,0.00030357452,0.000498366,0.003618428,0.0050119413,0.00083469733,0.0023747832,0.009399979],"category_scores_gemma":[0.01051979,0.00016779269,0.00026308978,0.00027505847,0.00536409,0.0020361983,0.0030420015,0.003318669,0.001416771],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024347343,0.00007263781,0.0030353647,0.00029367357,0.000009834698,0.0020833395,0.90999967,0.001105572,0.010989542,0.034124576,0.015676914,0.022365468],"study_design_scores_gemma":[0.00003404912,0.00033725434,0.004471673,0.0004509813,0.000021135027,0.0009761231,0.57388717,0.0042470302,0.006981345,0.012337773,0.39616045,0.00009497698],"about_ca_topic_score_codex":0.0016750324,"about_ca_topic_score_gemma":0.0041708485,"teacher_disagreement_score":0.009399979,"about_ca_system_score_codex":0.0017322229,"about_ca_system_score_gemma":0.00061965117,"threshold_uncertainty_score":0.03144604},"labels":[],"label_agreement":null},{"id":"W4407910063","doi":"10.2139/ssrn.5153814","title":"Tdri: Two-Phase Dialogue Refinement and Co-Adaptation for Interactive Image Generation","year":2025,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Adaptation (eye); Phase (matter); Image (mathematics); Computer science; Human–computer interaction; Artificial intelligence; Psychology; Chemistry","score_opus":0.02233089557443285,"score_gpt":0.35833504786809367,"score_spread":0.3360041522936608,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407910063","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004436807,0.00012088662,0.98616356,0.000045518416,0.000085081534,0.000120365265,0.000074632146,0.0075903074,0.0013628633],"genre_scores_gemma":[0.12816435,0.00010403093,0.8620226,0.00014574561,0.00007511025,0.00047519922,0.0005152845,0.0022856162,0.006212122],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99842876,0.0005209431,0.0000668476,0.00041998687,0.00041917077,0.00014419183],"domain_scores_gemma":[0.99789125,0.0011614714,0.0000642988,0.00042358853,0.00032854726,0.00013081564],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018474497,0.0017428254,0.0014376746,0.0008519746,0.0007096812,0.001470908,0.0035955426,0.0024159579,0.014461697],"category_scores_gemma":[0.0056466134,0.00083428336,0.0010761863,0.000779082,0.0007645586,0.0018812526,0.00477799,0.00244777,0.004971042],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012753545,0.00046613155,0.00043793247,0.0003859405,0.00013549454,0.00037906852,0.0009461997,0.05521228,0.09220362,0.008010304,0.013761907,0.82678586],"study_design_scores_gemma":[0.00012480785,0.0001611847,0.00026928153,0.000022228549,0.000033656706,0.000185635,0.00010222217,0.94790506,0.03842084,0.005406627,0.0073187174,0.00004970322],"about_ca_topic_score_codex":0.0034408122,"about_ca_topic_score_gemma":0.003619647,"teacher_disagreement_score":0.014461697,"about_ca_system_score_codex":0.0004238169,"about_ca_system_score_gemma":0.00073754316,"threshold_uncertainty_score":0.048379183},"labels":[],"label_agreement":null},{"id":"W4408014605","doi":"10.1016/j.eswa.2025.126965","title":"Referring Expression Comprehension in semi-structured human–robot interaction","year":2025,"lang":"en","type":"article","venue":"Expert Systems with Applications","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"Young Scientists Fund; National Natural Science Foundation of China","keywords":"Computer science; Expression (computer science); Human–robot interaction; Human–computer interaction; Artificial intelligence; Robot; Comprehension; Natural language processing; Programming language","score_opus":0.01705013750644576,"score_gpt":0.3216777303456205,"score_spread":0.3046275928391748,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408014605","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4103078,0.0008432426,0.56044537,0.00049641833,0.00012378601,0.00031161826,0.00047260296,0.003754941,0.02324427],"genre_scores_gemma":[0.9551744,0.0001687341,0.040941574,0.0000959941,0.000033429318,0.00013983373,0.00049029343,0.00025451957,0.002701229],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99670863,0.0022043881,0.00010344024,0.00048488975,0.0003618378,0.00013682267],"domain_scores_gemma":[0.9912845,0.0062753498,0.00069412816,0.0007167738,0.0008889663,0.00014030369],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018467824,0.0007223867,0.00056765374,0.00036971032,0.0005216052,0.001963224,0.0010674553,0.0014531479,0.007032926],"category_scores_gemma":[0.021170074,0.0005130896,0.00049441494,0.00044628663,0.0010681043,0.0038738123,0.0017418375,0.00092941505,0.0016973474],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004265593,0.00036521198,0.012106461,0.0023290413,0.00030689285,0.003971555,0.093242034,0.04998659,0.35811865,0.08760499,0.01346058,0.37424237],"study_design_scores_gemma":[0.00022705081,0.0015690243,0.04179871,0.00043655516,0.00027335653,0.0025500453,0.017740868,0.6520669,0.09380702,0.16250014,0.0266501,0.00038028933],"about_ca_topic_score_codex":0.001587912,"about_ca_topic_score_gemma":0.0007910041,"teacher_disagreement_score":0.007032926,"about_ca_system_score_codex":0.0004483034,"about_ca_system_score_gemma":0.0004978613,"threshold_uncertainty_score":0.023527503},"labels":[],"label_agreement":null},{"id":"W4408100255","doi":"10.1109/lra.2025.3547631","title":"GPT-Driven Gestures: Leveraging Large Language Models to Generate Expressive Robot Motion for Enhanced Human-Robot Interaction","year":2025,"lang":"en","type":"article","venue":"IEEE Robotics and Automation Letters","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary; University of Victoria","funders":"","keywords":"Gesture; Robot; Computer science; Motion (physics); Human–computer interaction; Human–robot interaction; Artificial intelligence; Communication; Psychology","score_opus":0.02232718226525392,"score_gpt":0.3102131524758498,"score_spread":0.28788597021059587,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408100255","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17765234,0.00032382482,0.8109036,0.00024750328,0.000087184286,0.00031236827,0.000472228,0.005322129,0.0046788445],"genre_scores_gemma":[0.7394166,0.00019173259,0.25446525,0.00020702116,0.000022467722,0.000441499,0.0008685949,0.0005219199,0.0038649475],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996798,0.00014456525,0.00001220546,0.0000743295,0.00006690123,0.000022352458],"domain_scores_gemma":[0.9994765,0.00035255656,0.000038394257,0.0000731107,0.000035180878,0.000024347199],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006674452,0.0007689023,0.00025862458,0.00018703063,0.00013974942,0.00039101555,0.00061863643,0.00042587504,0.0029704282],"category_scores_gemma":[0.0028335913,0.00021905615,0.00044112,0.00011546064,0.0004268576,0.0006405517,0.00096614746,0.0006499915,0.000980238],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005095313,0.00029747083,0.003630521,0.0006969161,0.00007989683,0.00080598803,0.0018308416,0.18473521,0.30640706,0.009474042,0.0054984135,0.48603413],"study_design_scores_gemma":[0.00006996256,0.0006703438,0.0025739102,0.00006262528,0.000041623127,0.000676384,0.00028053406,0.9261405,0.05406546,0.007122264,0.008232779,0.000063545274],"about_ca_topic_score_codex":0.00068419904,"about_ca_topic_score_gemma":0.0011080523,"teacher_disagreement_score":0.0029704282,"about_ca_system_score_codex":0.00020114699,"about_ca_system_score_gemma":0.0002712803,"threshold_uncertainty_score":0.009937048},"labels":[],"label_agreement":null},{"id":"W4408355766","doi":"10.1109/icassp49660.2025.10888770","title":"SX-Stitch: An Efficient VMS-UNet Based Framework for Intraoperative Scoliosis X-Ray Image Stitching","year":2025,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"National Natural Science Foundation of China","keywords":"Image stitching; Computer science; Scoliosis; Computer vision; Artificial intelligence; Surgery; Medicine","score_opus":0.013241337446606015,"score_gpt":0.33688605324469206,"score_spread":0.32364471579808607,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408355766","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008772451,0.00017642604,0.9879368,0.000051833256,0.000023923803,0.000053642943,0.00006333403,0.0023423529,0.0005791521],"genre_scores_gemma":[0.20618218,0.0002762261,0.7892344,0.00012363038,0.000042982618,0.00012781553,0.0004828437,0.00041175517,0.003118184],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99980396,0.000027946498,0.000010254454,0.000048042806,0.000084913285,0.000024969315],"domain_scores_gemma":[0.9998447,0.00003318715,0.000019721405,0.000039839815,0.00004312158,0.000019428135],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00031704176,0.00058774435,0.0005991759,0.00053810846,0.00026058813,0.0005224319,0.0011682705,0.0006864344,0.003210944],"category_scores_gemma":[0.00076288363,0.00038108625,0.0006942489,0.00032499712,0.00027151965,0.0005933744,0.0011283428,0.0009175005,0.00071542227],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030757886,0.00011290671,0.0013252288,0.000150943,0.00009838117,0.00021833935,0.0001393599,0.1683046,0.093953155,0.006525877,0.0043021976,0.72456145],"study_design_scores_gemma":[0.0000099733825,0.000067260306,0.0003828959,0.00000621745,0.000009539421,0.00015249428,0.000016574555,0.98274225,0.012120336,0.001646243,0.0028347948,0.000011444162],"about_ca_topic_score_codex":0.0032935368,"about_ca_topic_score_gemma":0.0071238265,"teacher_disagreement_score":0.0032935368,"about_ca_system_score_codex":0.00036780952,"about_ca_system_score_gemma":0.0008886432,"threshold_uncertainty_score":0.010741651},"labels":[],"label_agreement":null},{"id":"W4409263021","doi":"10.1109/wacv61041.2025.00265","title":"AiDe: Improving 3D Open-Vocabulary Semantic Segmentation by Aligned Vision-Language Learning","year":2025,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Natural language processing; Segmentation; Artificial intelligence; Vocabulary; Vocabulary learning; Image segmentation; Linguistics","score_opus":0.006589663231590595,"score_gpt":0.30603053788351264,"score_spread":0.29944087465192204,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409263021","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02082024,0.0007712982,0.94301355,0.000173364,0.000219156,0.00018594596,0.0017009105,0.029529994,0.0035855991],"genre_scores_gemma":[0.24290664,0.0005462373,0.72278345,0.0010043134,0.00013767571,0.0005997009,0.019865884,0.002594804,0.0095612565],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99873,0.00013904524,0.000057465833,0.0005800805,0.00033870197,0.00015460268],"domain_scores_gemma":[0.99922884,0.00018065692,0.000055150762,0.00026585557,0.00019512408,0.00007433721],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007825024,0.0028725648,0.0021470902,0.0019167664,0.0007022203,0.0019096615,0.0040704818,0.002303368,0.0066902586],"category_scores_gemma":[0.0027668194,0.0009668542,0.002518901,0.0016802186,0.0012737402,0.0048134793,0.0041036145,0.0024389324,0.0057960134],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005146621,0.0004556215,0.0013584702,0.00048782513,0.00023278862,0.00051070424,0.0004313197,0.11872489,0.062015753,0.01263905,0.033623777,0.76900524],"study_design_scores_gemma":[0.00006155825,0.00015008089,0.00047129017,0.000033152777,0.00004183401,0.00023313698,0.00014220829,0.94342047,0.02727228,0.017943693,0.010170582,0.000059774113],"about_ca_topic_score_codex":0.0087674195,"about_ca_topic_score_gemma":0.013198052,"teacher_disagreement_score":0.0087674195,"about_ca_system_score_codex":0.0011528141,"about_ca_system_score_gemma":0.0016343296,"threshold_uncertainty_score":0.022381186},"labels":[],"label_agreement":null},{"id":"W4409362862","doi":"10.1609/aaai.v39i25.34900","title":"Learn2Aggregate: Supervised Generation of Chvatal-Gomory Cuts Using Graph Neural Networks","year":2025,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Toronto","funders":"","keywords":"Graph; Mathematics; Artificial neural network; Artificial intelligence; Computer science; Combinatorics; Pattern recognition (psychology)","score_opus":0.09341661046799989,"score_gpt":0.3222071388322298,"score_spread":0.22879052836422992,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409362862","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023709582,0.00025387044,0.965556,0.00027205277,0.00009018253,0.00016890561,0.0003216473,0.005229801,0.0043979613],"genre_scores_gemma":[0.25798342,0.000116632065,0.7340687,0.00038137616,0.000045806555,0.00037285069,0.0015057849,0.0018842518,0.0036412522],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99948335,0.0001360847,0.000018987388,0.00013124621,0.0001579089,0.000072501294],"domain_scores_gemma":[0.9989249,0.0005905984,0.00009180095,0.0001622411,0.00016560423,0.00006476273],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009791721,0.0016991679,0.0009922273,0.00090512563,0.00055225147,0.001151789,0.0017409471,0.0016081173,0.0071019907],"category_scores_gemma":[0.0033761184,0.0006707298,0.0009860896,0.0006091543,0.000952931,0.0012196772,0.0016339553,0.0025203258,0.001285483],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016562674,0.00012655953,0.0011298475,0.00018346695,0.00006177709,0.00014876775,0.00008519086,0.7780841,0.004460869,0.017304184,0.01210558,0.18614393],"study_design_scores_gemma":[0.000023239394,0.000034352808,0.000054555716,0.00001130625,0.0000050609588,0.000018868268,0.000014040841,0.9896172,0.0014461286,0.0073546306,0.0014155967,0.0000049176147],"about_ca_topic_score_codex":0.004073514,"about_ca_topic_score_gemma":0.010865168,"teacher_disagreement_score":0.0071019907,"about_ca_system_score_codex":0.0010279783,"about_ca_system_score_gemma":0.0016786893,"threshold_uncertainty_score":0.023758471},"labels":[],"label_agreement":null},{"id":"W4409364374","doi":"10.1609/aaai.v39i16.33918","title":"Enhance Vision-Language Alignment with Noise","year":2025,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Noise (video); Computer vision; Artificial intelligence; Image (mathematics)","score_opus":0.019751031522510835,"score_gpt":0.3212355657977287,"score_spread":0.3014845342752179,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409364374","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04987692,0.0007105251,0.9419777,0.00054656994,0.00018856113,0.0000977899,0.00022015678,0.0035834508,0.0027983163],"genre_scores_gemma":[0.74129456,0.00042198386,0.24640875,0.0014670885,0.00023375513,0.00026851808,0.0013459532,0.0007637276,0.0077956654],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989761,0.00023290368,0.000037096837,0.00041895005,0.00019908507,0.00013579115],"domain_scores_gemma":[0.9984597,0.0007048565,0.00014053035,0.00030785147,0.00026419593,0.0001227488],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019883665,0.0017960969,0.001176877,0.00065147685,0.00062535115,0.0013673401,0.0022182232,0.0017263726,0.002442104],"category_scores_gemma":[0.010215157,0.0005683072,0.00092174317,0.0005513503,0.0012593194,0.0034052413,0.003060539,0.0031465625,0.0010955104],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000785639,0.0006007099,0.0045637405,0.0003518442,0.00023750344,0.00032357892,0.0005140876,0.5083686,0.060013566,0.025572062,0.009892175,0.38877645],"study_design_scores_gemma":[0.000020455187,0.0000743781,0.0002953796,0.000014327998,0.000022157166,0.00004964303,0.000027298784,0.9787472,0.010736821,0.008918301,0.001078359,0.000015704416],"about_ca_topic_score_codex":0.004608662,"about_ca_topic_score_gemma":0.0053695575,"teacher_disagreement_score":0.004608662,"about_ca_system_score_codex":0.0010469811,"about_ca_system_score_gemma":0.0014313605,"threshold_uncertainty_score":0.01051563},"labels":[],"label_agreement":null},{"id":"W4409439325","doi":"10.3390/ijgi14040170","title":"Text Geolocation Prediction via Self-Supervised Learning","year":2025,"lang":"en","type":"article","venue":"ISPRS International Journal of Geo-Information","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"National Natural Science Foundation of China","keywords":"Geolocation; Computer science; Artificial intelligence; Machine learning; Data science; Information retrieval; World Wide Web","score_opus":0.0027755090878996986,"score_gpt":0.24370838338605022,"score_spread":0.24093287429815052,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409439325","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14338486,0.001125656,0.832569,0.0010323055,0.00045095547,0.00020156673,0.0061050565,0.009043446,0.0060872086],"genre_scores_gemma":[0.81494105,0.00045109622,0.15948497,0.00034444843,0.00035331966,0.00020640677,0.014126835,0.00034210476,0.0097497245],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994848,0.00010320774,0.00002909598,0.0002634527,0.00007389675,0.000045561606],"domain_scores_gemma":[0.99890435,0.0003803728,0.00019930366,0.00022033844,0.00023810615,0.00005741007],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004623111,0.0013195379,0.000671385,0.0019008677,0.00042726984,0.0005797371,0.0017912894,0.0009794547,0.0018102868],"category_scores_gemma":[0.0027803301,0.00026174024,0.00052690005,0.0018733819,0.00056074787,0.0024335715,0.0010665049,0.0010807501,0.002226447],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006373561,0.0004215948,0.02351347,0.00040775636,0.0001868423,0.00050656026,0.0004663466,0.27189183,0.01160709,0.0058465004,0.049007762,0.63550687],"study_design_scores_gemma":[0.000022073784,0.000045530003,0.002075793,0.000022745873,0.000028257311,0.00009076804,0.00012415432,0.9811569,0.0039267344,0.008228721,0.004262554,0.000015730817],"about_ca_topic_score_codex":0.0039365045,"about_ca_topic_score_gemma":0.008563066,"teacher_disagreement_score":0.0039365045,"about_ca_system_score_codex":0.00050928077,"about_ca_system_score_gemma":0.00051824417,"threshold_uncertainty_score":0.007827163},"labels":[],"label_agreement":null},{"id":"W4409539127","doi":"10.1145/3729242","title":"Cascade Transformer for Hierarchical Semantic Reasoning in Text-Based Visual Question Answering","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Intelligent Systems and Technology","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"St. Francis Xavier University","funders":"National Natural Science Foundation of China","keywords":"Computer science; Question answering; Transformer; Cascade; Natural language processing; Artificial intelligence; Information retrieval","score_opus":0.010491793156558554,"score_gpt":0.308018523082175,"score_spread":0.2975267299256164,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409539127","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020684157,0.00089852,0.9647592,0.00038607678,0.00010702123,0.00030727359,0.00080508715,0.008257406,0.003795258],"genre_scores_gemma":[0.5885856,0.0008149458,0.39742932,0.00082338677,0.000117670235,0.00038869842,0.0036946372,0.00031215852,0.007833611],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9993399,0.000108739136,0.000042268926,0.00028931277,0.00014027073,0.00007959183],"domain_scores_gemma":[0.99944,0.0002552972,0.000042311192,0.00008778939,0.00013341969,0.000041188065],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010450756,0.0012066017,0.0007105925,0.0012329193,0.00042228037,0.001006461,0.0024381552,0.0013204966,0.0066262186],"category_scores_gemma":[0.002895039,0.00044569594,0.0018628688,0.0007225037,0.0007091211,0.00354756,0.0014494902,0.0017096538,0.0020159734],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00073647866,0.0005204321,0.0018497078,0.0007295528,0.00018507798,0.00065697776,0.0010064262,0.087279074,0.059120964,0.029570043,0.02054867,0.7977966],"study_design_scores_gemma":[0.000040821436,0.00011728951,0.0006012848,0.0000394933,0.00010272467,0.00019683647,0.00012977357,0.92489654,0.018189339,0.04944407,0.0062063555,0.000035475445],"about_ca_topic_score_codex":0.010348385,"about_ca_topic_score_gemma":0.013198014,"teacher_disagreement_score":0.010348385,"about_ca_system_score_codex":0.001466869,"about_ca_system_score_gemma":0.0010958988,"threshold_uncertainty_score":0.022166908},"labels":[],"label_agreement":null},{"id":"W4410196438","doi":"10.1007/s44196-025-00853-0","title":"Dual Adapter Tuning of Vision–Language Models Using Large Language Models","year":2025,"lang":"en","type":"article","venue":"International Journal of Computational Intelligence Systems","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada; Alliance de recherche numérique du Canada","keywords":"Computer science; Adapter (computing); Language model; Dual (grammatical number); Artificial intelligence; Natural language processing; Linguistics; Computer hardware","score_opus":0.029692916415672392,"score_gpt":0.364484520778666,"score_spread":0.3347916043629936,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410196438","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09157837,0.0020669748,0.84852153,0.0006225491,0.00044312034,0.00040010124,0.0015783075,0.05042602,0.0043630307],"genre_scores_gemma":[0.6450903,0.00067310326,0.33292228,0.001166729,0.00019512462,0.0009400308,0.007292742,0.0015652163,0.010154595],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99892056,0.0002511324,0.00006425454,0.00052370265,0.00012172074,0.00011862427],"domain_scores_gemma":[0.9975574,0.0013195777,0.00011091598,0.0005056934,0.0003811292,0.0001252602],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022374804,0.0023293255,0.0013763084,0.0009701808,0.00049033243,0.0017594812,0.0038060953,0.0021509123,0.005087364],"category_scores_gemma":[0.00786747,0.0008879289,0.0018414248,0.0009981666,0.0007601405,0.0041481657,0.0023742095,0.004183932,0.0040633027],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059040205,0.0005901561,0.0022505438,0.00034567295,0.00025460764,0.00028830097,0.00025866088,0.25994882,0.013822142,0.004197726,0.016825275,0.7006277],"study_design_scores_gemma":[0.00004581061,0.00009307821,0.00024258257,0.00001721659,0.000027934144,0.000059174326,0.00004473951,0.988488,0.0047727977,0.0046220287,0.0015648012,0.000021825726],"about_ca_topic_score_codex":0.008564398,"about_ca_topic_score_gemma":0.009519652,"teacher_disagreement_score":0.008564398,"about_ca_system_score_codex":0.0016093672,"about_ca_system_score_gemma":0.0013463245,"threshold_uncertainty_score":0.017029107},"labels":[],"label_agreement":null},{"id":"W4410773142","doi":"10.1016/j.neucom.2025.130504","title":"Synergy-driven multi-modal prompting for weakly supervised semantic segmentation","year":2025,"lang":"en","type":"article","venue":"Neurocomputing","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"Key Technology Research and Development Program of Shandong; National Natural Science Foundation of China","keywords":"Computer science; Segmentation; Modal; Artificial intelligence; Natural language processing; Pattern recognition (psychology); Computer vision","score_opus":0.01945021987538133,"score_gpt":0.3093893573511156,"score_spread":0.2899391374757343,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410773142","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01555666,0.00034674874,0.9744409,0.00022889588,0.00013893409,0.000104643565,0.00044565412,0.0065288036,0.0022088825],"genre_scores_gemma":[0.4389504,0.00038285516,0.55003905,0.0004439493,0.00018624443,0.00032364565,0.0021685238,0.0013395873,0.006165775],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9991866,0.00019846915,0.0000450739,0.0002872813,0.00016088883,0.00012172216],"domain_scores_gemma":[0.997982,0.0010743983,0.000111666166,0.00028759902,0.00036961283,0.00017464803],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013940212,0.0014330101,0.001369514,0.0010264169,0.00078170537,0.0012578677,0.0020777371,0.0020708633,0.009001689],"category_scores_gemma":[0.0049707284,0.00063889846,0.0009661503,0.0011978585,0.00071453105,0.0023288392,0.0035806613,0.0024981461,0.0034906352],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025611573,0.00040710892,0.001160806,0.0005624319,0.00009691181,0.0005684823,0.00051690417,0.0802582,0.104279764,0.014008993,0.015870612,0.7797087],"study_design_scores_gemma":[0.000041210176,0.00014644397,0.00048581703,0.000037674323,0.00003356488,0.00015553403,0.00012693228,0.93529236,0.026551647,0.032704175,0.0043932945,0.000031322925],"about_ca_topic_score_codex":0.002159814,"about_ca_topic_score_gemma":0.0048712944,"teacher_disagreement_score":0.009001689,"about_ca_system_score_codex":0.00060286414,"about_ca_system_score_gemma":0.001635666,"threshold_uncertainty_score":0.030113697},"labels":[],"label_agreement":null},{"id":"W4410970740","doi":"10.1007/s13735-025-00370-y","title":"Chameleon: A Multimodal Learning Framework Robust to Missing Modalities","year":2025,"lang":"en","type":"article","venue":"International Journal of Multimedia Information Retrieval","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"Austrian Science Fund","keywords":"Computer science; Modalities; Artificial intelligence; Machine learning","score_opus":0.009476488817652846,"score_gpt":0.2941190413649713,"score_spread":0.28464255254731846,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410970740","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018503934,0.0010406176,0.97064203,0.00046413593,0.00011128188,0.00011835054,0.00023008452,0.005024648,0.0038648918],"genre_scores_gemma":[0.5641966,0.0005858147,0.4202741,0.00073602534,0.00021881092,0.00038347216,0.0009183381,0.0006313668,0.012055468],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99929,0.00021485116,0.00002100863,0.00020163467,0.00017536727,0.0000972039],"domain_scores_gemma":[0.9995034,0.00016907595,0.00004453519,0.00009433587,0.00013043478,0.00005817038],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017878533,0.0010646147,0.00097351824,0.0010558652,0.0005870747,0.00093197194,0.002476114,0.0014738104,0.006859622],"category_scores_gemma":[0.0025566309,0.00032471374,0.00077387405,0.00060138607,0.00096006563,0.0014986286,0.0029763016,0.001740132,0.0013387722],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00073248334,0.00025579424,0.00079358957,0.00021968853,0.00018249122,0.0002672884,0.00015328104,0.25850782,0.032425415,0.021245662,0.014929414,0.67028713],"study_design_scores_gemma":[0.00002768158,0.00010117741,0.00018778793,0.000017641118,0.000019292283,0.000057531503,0.0000237793,0.97803295,0.0053357645,0.013603293,0.0025756308,0.000017457738],"about_ca_topic_score_codex":0.0041423077,"about_ca_topic_score_gemma":0.0056342706,"teacher_disagreement_score":0.006859622,"about_ca_system_score_codex":0.00078430714,"about_ca_system_score_gemma":0.0010113631,"threshold_uncertainty_score":0.022947729},"labels":[],"label_agreement":null},{"id":"W4411119193","doi":"10.18653/v1/2025.naacl-srw.37","title":"ELIOT: Zero-Shot Video-Text Retrieval through Relevance-Boosted Captioning and Structural Information Extraction","year":2025,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Closed captioning; Shot (pellet); Relevance (law); Computer science; Zero (linguistics); Information retrieval; Feature extraction; Artificial intelligence; Extraction (chemistry); Natural language processing; Speech recognition; Image (mathematics); Linguistics","score_opus":0.012664193052196262,"score_gpt":0.30361140910088824,"score_spread":0.290947216048692,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411119193","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029355079,0.01350659,0.80686396,0.0012146661,0.0037310978,0.0011632071,0.017638735,0.11175838,0.014768322],"genre_scores_gemma":[0.10312751,0.0032966877,0.78587836,0.00086949184,0.001463211,0.0007757827,0.064311564,0.0036305645,0.036646806],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99884987,0.00018778366,0.000080494916,0.0003320126,0.00038052004,0.00016927137],"domain_scores_gemma":[0.998566,0.00038745673,0.00009975968,0.00034436723,0.00049235293,0.00011017838],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011884628,0.0029122834,0.0023774935,0.0063538547,0.0010950603,0.0022878458,0.003287027,0.0023630362,0.017383547],"category_scores_gemma":[0.002937487,0.0006240065,0.0013405624,0.0036174192,0.0007266554,0.0038855302,0.002519591,0.0015401754,0.017751234],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008604269,0.000353106,0.00031974635,0.0007176812,0.00018264525,0.00038873975,0.00012163954,0.0026289907,0.08146675,0.0020077263,0.18550047,0.72545207],"study_design_scores_gemma":[0.0007215926,0.0011544706,0.0034257416,0.00017191416,0.00053677784,0.0016154327,0.00054367055,0.660489,0.19460543,0.014496366,0.12191811,0.0003215111],"about_ca_topic_score_codex":0.0066515394,"about_ca_topic_score_gemma":0.011103715,"teacher_disagreement_score":0.017383547,"about_ca_system_score_codex":0.00077905564,"about_ca_system_score_gemma":0.0012868409,"threshold_uncertainty_score":0.05815375},"labels":[],"label_agreement":null},{"id":"W4411176006","doi":"10.1016/j.nlp.2025.100159","title":"Next-generation image captioning: A survey of methodologies and emerging challenges from transformers to Multimodal Large Language Models","year":2025,"lang":"en","type":"article","venue":"Natural Language Processing Journal","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Closed captioning; Computer science; Transformer; Natural language processing; Artificial intelligence; Image (mathematics); Engineering; Electrical engineering","score_opus":0.0630583087593844,"score_gpt":0.36922933374050865,"score_spread":0.30617102498112425,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411176006","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0054418324,0.04114411,0.92937875,0.0030813825,0.0005160657,0.00022774238,0.0008695492,0.007503779,0.0118367635],"genre_scores_gemma":[0.17809375,0.07177547,0.7191174,0.0028580022,0.001546573,0.0005961126,0.006533983,0.0029917294,0.016486965],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99880683,0.00043488792,0.00009053437,0.00026640636,0.00032858195,0.00007271207],"domain_scores_gemma":[0.99718815,0.0016159638,0.00012202989,0.00044566795,0.00053720764,0.000090991765],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002763933,0.0018039485,0.0012502279,0.002030037,0.0005468454,0.0033187398,0.0029198118,0.0015798174,0.008154644],"category_scores_gemma":[0.007123054,0.0006499955,0.0012108439,0.002133507,0.0011728476,0.006005218,0.002399155,0.0029337157,0.0044379705],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014466593,0.00013461502,0.00044277948,0.0014904654,0.00009842421,0.00019335213,0.00035178658,0.059276856,0.0062452373,0.052819874,0.03720463,0.8415973],"study_design_scores_gemma":[0.000028081775,0.00017491406,0.00034671853,0.00046832874,0.00007368151,0.00062535336,0.00027562745,0.795644,0.014261075,0.08294562,0.105060406,0.00009622068],"about_ca_topic_score_codex":0.004702269,"about_ca_topic_score_gemma":0.004344603,"teacher_disagreement_score":0.008154644,"about_ca_system_score_codex":0.0018008668,"about_ca_system_score_gemma":0.0012289897,"threshold_uncertainty_score":0.027279973},"labels":[],"label_agreement":null},{"id":"W4411357823","doi":"10.1007/978-3-031-82606-1_2","title":"Navigational Assistance for the Blind in Complex Indoor Spaces Using a Vision-Enabled Large Language Model","year":2025,"lang":"en","type":"book-chapter","venue":"Studies in computational intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Thompson Rivers University; Algoma University","funders":"","keywords":"Computer science; Artificial intelligence; Computer vision; Human–computer interaction","score_opus":0.11730223578359025,"score_gpt":0.43918222356397013,"score_spread":0.32187998778037985,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411357823","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021368193,0.0003756211,0.9700636,0.00023047147,0.00009692034,0.000019053103,0.00018941186,0.0028811393,0.0047754254],"genre_scores_gemma":[0.63243705,0.0006796433,0.3477122,0.00017604606,0.000068962065,0.00007093503,0.0005462801,0.00037797843,0.017930858],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99991214,0.000016795233,0.0000044561034,0.000025897567,0.000024262456,0.000016479622],"domain_scores_gemma":[0.99986696,0.00006758033,0.0000076253637,0.000021710752,0.0000245554,0.000011533023],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00017233542,0.00040369006,0.00050162064,0.00016226026,0.0003090388,0.00075237604,0.00065866986,0.0005419573,0.0029984303],"category_scores_gemma":[0.0005693164,0.00020999218,0.0006567351,0.00026086727,0.00034889803,0.0012291237,0.0009441857,0.00089520693,0.000887835],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037545618,0.00020802813,0.0010185625,0.00023942383,0.00012811291,0.0005143283,0.00054007815,0.40377253,0.051777028,0.044904143,0.020258151,0.4762641],"study_design_scores_gemma":[0.000007517772,0.00002777651,0.00016995196,0.000008204874,0.000014880106,0.00006467786,0.000051586834,0.9830487,0.0030592578,0.011185739,0.00234814,0.000013627486],"about_ca_topic_score_codex":0.010608949,"about_ca_topic_score_gemma":0.016948748,"teacher_disagreement_score":0.010608949,"about_ca_system_score_codex":0.00027587923,"about_ca_system_score_gemma":0.00075551,"threshold_uncertainty_score":0.021094382},"labels":[],"label_agreement":null},{"id":"W4411449683","doi":"10.1145/3729343","title":"VLATest: Testing and Evaluating Vision-Language-Action Models for Robotic Manipulation","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Robustness (evolution); Artificial intelligence; Software deployment; Machine learning; Generative grammar; Human–computer interaction; Software engineering","score_opus":0.04418603467234083,"score_gpt":0.3291355321729803,"score_spread":0.28494949750063947,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411449683","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6558537,0.00097640714,0.31557435,0.0006797121,0.00034323326,0.0012077261,0.001987693,0.016430315,0.006946956],"genre_scores_gemma":[0.83139485,0.0002398218,0.16296664,0.0002470521,0.000026491367,0.0006390127,0.0024863163,0.00062987837,0.0013699959],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9983253,0.00057344924,0.0001157777,0.00040698075,0.00044464954,0.00013387788],"domain_scores_gemma":[0.9915685,0.0064569917,0.0004107166,0.00083907763,0.000481562,0.00024310288],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032024148,0.0016748578,0.0006235217,0.000867647,0.0005213654,0.0009809466,0.0034144383,0.0021770254,0.0030363582],"category_scores_gemma":[0.013307004,0.0007530313,0.0013651451,0.0003395785,0.0015721455,0.0019075904,0.0019317155,0.0021470238,0.0006799424],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00073347037,0.00085010973,0.0057362136,0.00061672984,0.00020568751,0.00019154913,0.00026245785,0.8998533,0.012455232,0.0050714584,0.0040458185,0.069977894],"study_design_scores_gemma":[0.000049612237,0.00028203265,0.0005378685,0.000019749219,0.000014181999,0.000034756824,0.00002699867,0.9930728,0.004365382,0.0010002931,0.0005819615,0.0000142949475],"about_ca_topic_score_codex":0.01230621,"about_ca_topic_score_gemma":0.012434086,"teacher_disagreement_score":0.01230621,"about_ca_system_score_codex":0.0019022903,"about_ca_system_score_gemma":0.0016691722,"threshold_uncertainty_score":0.024469137},"labels":[],"label_agreement":null},{"id":"W4412106036","doi":"10.1016/j.cviu.2025.104442","title":"FedVLP: Visual-aware latent prompt generation for Multimodal Federated Learning","year":2025,"lang":"en","type":"article","venue":"Computer Vision and Image Understanding","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"National Natural Science Foundation of China","keywords":"Computer science; Artificial intelligence; Machine learning","score_opus":0.032529756933271155,"score_gpt":0.3306209197960363,"score_spread":0.29809116286276516,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412106036","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0059926305,0.00031062507,0.9190673,0.00018708638,0.0002173925,0.00021254666,0.002193066,0.069867045,0.0019522037],"genre_scores_gemma":[0.20271803,0.0002925967,0.774361,0.0004910559,0.00012055712,0.00088769104,0.007976163,0.003117609,0.010035334],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992077,0.00020687323,0.000037603273,0.0002630932,0.0001663115,0.00011850355],"domain_scores_gemma":[0.99875045,0.0005449095,0.00005062163,0.00033153757,0.00022743299,0.00009509641],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012739191,0.0020435106,0.0012378396,0.0012785754,0.0007835959,0.0015163905,0.0025984612,0.0021453889,0.030782307],"category_scores_gemma":[0.0057364767,0.00059996144,0.0014133643,0.0010185476,0.0005944006,0.0026879816,0.004997874,0.0024717904,0.008932672],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011260229,0.00042204582,0.0007632834,0.00031747125,0.00009448614,0.0002847496,0.00022646924,0.02621666,0.018175788,0.0061045554,0.05460609,0.89166236],"study_design_scores_gemma":[0.0001348399,0.00017278882,0.00041494577,0.00006229325,0.000040603165,0.0001715837,0.00012804671,0.9262588,0.028615173,0.029784594,0.014161968,0.000054466265],"about_ca_topic_score_codex":0.00539458,"about_ca_topic_score_gemma":0.006903201,"teacher_disagreement_score":0.030782307,"about_ca_system_score_codex":0.001007059,"about_ca_system_score_gemma":0.0013349933,"threshold_uncertainty_score":0.1029771},"labels":[],"label_agreement":null},{"id":"W4413038007","doi":"10.1038/s42256-025-01072-0","title":"High-level visual representations in the human brain are aligned with large language models","year":2025,"lang":"en","type":"article","venue":"Nature Machine Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; National Institutes of Health; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; National Science Foundation","keywords":"Computer science; Artificial intelligence; Brain activity and meditation; Natural language processing; Cognitive psychology; Psychology; Neuroscience; Electroencephalography","score_opus":0.01296838644040375,"score_gpt":0.3531533419880927,"score_spread":0.3401849555476889,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413038007","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2437214,0.0009658233,0.742696,0.0015396399,0.0001329926,0.00007636897,0.001155067,0.0019794723,0.007733284],"genre_scores_gemma":[0.94493794,0.0005351084,0.051206596,0.00022398171,0.000045496872,0.000082036415,0.0008604097,0.00019444266,0.0019140585],"study_design_codex":"simulation_or_modeling","study_design_gemma":"observational","domain_scores_codex":[0.9996705,0.00012219737,0.000012839565,0.00011650283,0.000050913604,0.00002699855],"domain_scores_gemma":[0.9991289,0.00044482117,0.00013518069,0.00015785123,0.000098268414,0.000035054258],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00051388616,0.00072334695,0.00032251,0.00040562174,0.0001850734,0.0012322834,0.0005675674,0.0005813811,0.0023280415],"category_scores_gemma":[0.00526648,0.00038082548,0.00073136535,0.000407483,0.0007357004,0.0025074051,0.00077046274,0.0013527052,0.0008368822],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003992958,0.00016477262,0.006410772,0.00052682473,0.00035604768,0.0002845973,0.00082835817,0.56918114,0.11611024,0.046819415,0.006562455,0.252356],"study_design_scores_gemma":[0.000010014199,0.000086113236,0.0059087556,0.000036839356,0.00003132885,0.000106421576,0.00012382791,0.92542815,0.008627246,0.05789202,0.0017165535,0.00003262121],"about_ca_topic_score_codex":0.0020667876,"about_ca_topic_score_gemma":0.0026688157,"teacher_disagreement_score":0.0023280415,"about_ca_system_score_codex":0.0005696974,"about_ca_system_score_gemma":0.00034658733,"threshold_uncertainty_score":0.007788062},"labels":[],"label_agreement":null},{"id":"W4413144748","doi":"10.1109/cvpr52734.2025.01367","title":"Can Large Vision-Language Models Correct Semantic Grounding Errors By Themselves?","year":2025,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute","funders":"","keywords":"Computer science; Natural language processing; Ground; Artificial intelligence; Engineering","score_opus":0.0074882238277879185,"score_gpt":0.2949452169845906,"score_spread":0.2874569931568027,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413144748","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20013632,0.0016529664,0.7444264,0.0034045775,0.0007172232,0.00023912526,0.0006782951,0.03990692,0.008838199],"genre_scores_gemma":[0.82796323,0.00022834231,0.16429442,0.0011345014,0.000073042815,0.0001125114,0.0010217706,0.0018097135,0.0033624782],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99761534,0.0008382555,0.00009638449,0.0008573971,0.0003149209,0.00027775427],"domain_scores_gemma":[0.99118364,0.0044162893,0.000511484,0.0025933615,0.0009901744,0.0003050051],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044465256,0.0027136263,0.0016133416,0.00073418865,0.00083059183,0.0030992958,0.0045516044,0.002802928,0.0042632376],"category_scores_gemma":[0.031231515,0.0010782311,0.0012569022,0.0005946105,0.0017656292,0.007541247,0.0034653035,0.0054619834,0.0031212396],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008562849,0.0004465976,0.010997989,0.0005044751,0.0004788194,0.00046734477,0.0010119843,0.3627898,0.029989833,0.011143459,0.017042708,0.5642706],"study_design_scores_gemma":[0.00006683688,0.00015776377,0.0008316053,0.000063539555,0.00006779301,0.0001495934,0.00025669864,0.95176464,0.01587198,0.027117586,0.0036054796,0.00004653754],"about_ca_topic_score_codex":0.008863611,"about_ca_topic_score_gemma":0.0127341235,"teacher_disagreement_score":0.008863611,"about_ca_system_score_codex":0.0013740571,"about_ca_system_score_gemma":0.0023335202,"threshold_uncertainty_score":0.02351576},"labels":[],"label_agreement":null},{"id":"W4413145524","doi":"10.1109/cvpr52734.2025.01376","title":"BiomedCoOp: Learning to Prompt for Biomedical Vision-Language Models","year":2025,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Artificial intelligence; Natural language processing","score_opus":0.00969014896680131,"score_gpt":0.3362562618677056,"score_spread":0.32656611290090426,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413145524","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009641281,0.0007616312,0.96040374,0.000543303,0.00022187656,0.00021301187,0.0010207767,0.026172262,0.0010220865],"genre_scores_gemma":[0.28994682,0.0010030187,0.68896353,0.0024524783,0.00041698827,0.0012699318,0.0074866693,0.0025037413,0.005956874],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9987237,0.00036845426,0.0000627152,0.0005301159,0.00019913614,0.00011584544],"domain_scores_gemma":[0.9980909,0.00095032394,0.00011305083,0.00032335366,0.00035572195,0.00016660032],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024657042,0.0026193908,0.0013594624,0.0010945648,0.0007103172,0.0015937808,0.0030344143,0.002345007,0.0054642814],"category_scores_gemma":[0.010848802,0.0009397115,0.0016370277,0.00087299285,0.0010442776,0.003207811,0.0042509967,0.005389064,0.0032660628],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007396159,0.0005005704,0.0031137834,0.00058879686,0.0001839777,0.00036009768,0.00036172263,0.17032899,0.019793818,0.015787154,0.05078371,0.73745763],"study_design_scores_gemma":[0.000067033594,0.00016966819,0.0002994149,0.00003664345,0.0000276078,0.000102918966,0.00008168049,0.95752996,0.0069911014,0.028817466,0.0058405492,0.00003595054],"about_ca_topic_score_codex":0.004046582,"about_ca_topic_score_gemma":0.0077816374,"teacher_disagreement_score":0.0054642814,"about_ca_system_score_codex":0.0012506585,"about_ca_system_score_gemma":0.0030938585,"threshold_uncertainty_score":0.01827985},"labels":[],"label_agreement":null},{"id":"W4413145832","doi":"10.1109/cvpr52734.2025.02749","title":"CTRL-O: Language-Controllable Object-Centric Visual Representation Learning","year":2025,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Computer science; Representation (politics); Artificial intelligence; Object (grammar); Natural language processing; Programming language","score_opus":0.005370343066093995,"score_gpt":0.31150685231513425,"score_spread":0.30613650924904023,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413145832","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010209534,0.0002250154,0.9765286,0.0002341049,0.000044862314,0.00013453954,0.00041686292,0.010435896,0.0017707021],"genre_scores_gemma":[0.40111825,0.00039316897,0.58623487,0.0009943121,0.00007710645,0.00076038146,0.0028687287,0.001344457,0.006208643],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9989704,0.00025458654,0.000046648212,0.00043020688,0.00018836165,0.00010971186],"domain_scores_gemma":[0.99824655,0.0007520901,0.00014680339,0.00055768056,0.00018088978,0.000116000214],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013372208,0.0013500666,0.0008599389,0.00059687806,0.00038276982,0.0013878259,0.0048632156,0.0014322726,0.006109294],"category_scores_gemma":[0.005470377,0.0006829502,0.0012474295,0.00078565127,0.0014021938,0.0041143056,0.0037869485,0.002722396,0.0019794793],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006758871,0.0006959973,0.0017955607,0.000644242,0.00012918244,0.00027983246,0.0005563077,0.14992335,0.041965608,0.051378045,0.03140356,0.7205525],"study_design_scores_gemma":[0.000060236333,0.00013366576,0.0001459268,0.000017320637,0.000016425895,0.00007214345,0.00004903095,0.95309323,0.011876784,0.03002811,0.0044768085,0.000030244824],"about_ca_topic_score_codex":0.0041365894,"about_ca_topic_score_gemma":0.00591397,"teacher_disagreement_score":0.006109294,"about_ca_system_score_codex":0.0012064776,"about_ca_system_score_gemma":0.0013178887,"threshold_uncertainty_score":0.020437658},"labels":[],"label_agreement":null},{"id":"W4413145840","doi":"10.1109/cvpr52734.2025.00877","title":"DivPrune: Diversity-based Visual Token Pruning for Large Multimodal Models","year":2025,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada)","funders":"","keywords":"Computer science; Security token; Pruning; Diversity (politics); Artificial intelligence; Computer network; Biology","score_opus":0.01616403560729983,"score_gpt":0.3084121409683172,"score_spread":0.29224810536101736,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413145840","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025946302,0.0008655465,0.9595418,0.0003324185,0.000085151856,0.00017588954,0.000723884,0.010527971,0.00180115],"genre_scores_gemma":[0.3166347,0.00037216287,0.671697,0.0004336646,0.000085414475,0.0003834372,0.0034473792,0.0018919305,0.0050542424],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9991949,0.00020280338,0.000052044892,0.00024118158,0.00018702394,0.00012205039],"domain_scores_gemma":[0.99812526,0.0011048395,0.00014352167,0.00030769056,0.00021638494,0.000102295715],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015956857,0.0019781638,0.001693859,0.0010871202,0.00088492915,0.0015435044,0.0035781327,0.0017644597,0.0050396225],"category_scores_gemma":[0.0056763557,0.0009969035,0.0016748554,0.00087903853,0.0009673568,0.0024890916,0.0024493798,0.0024427895,0.001638151],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00060028984,0.00019559845,0.0025756443,0.0004135726,0.00023904769,0.00062623504,0.0002921099,0.5321742,0.012312211,0.011327925,0.019786289,0.41945687],"study_design_scores_gemma":[0.000023371009,0.000033321096,0.00010850752,0.000014678217,0.0000133635085,0.00008636639,0.000024998915,0.98864913,0.0035259116,0.0059703845,0.0015404344,0.000009469155],"about_ca_topic_score_codex":0.010482259,"about_ca_topic_score_gemma":0.019390183,"teacher_disagreement_score":0.010482259,"about_ca_system_score_codex":0.001487329,"about_ca_system_score_gemma":0.0018956953,"threshold_uncertainty_score":0.020842493},"labels":[],"label_agreement":null},{"id":"W4413146224","doi":"10.1109/cvpr52734.2025.02140","title":"VidSeg: Training-free Video Semantic Segmentation based on Diffusion Models","year":2025,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Computer science; Training (meteorology); Segmentation; Artificial intelligence; Diffusion; Image segmentation; Computer vision","score_opus":0.020627162872346897,"score_gpt":0.28407019326650984,"score_spread":0.26344303039416295,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413146224","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011751535,0.0004486589,0.97925925,0.00018747654,0.000104101906,0.00008129933,0.00037002488,0.0064798263,0.0013177088],"genre_scores_gemma":[0.2816448,0.00065437565,0.70575964,0.00041717122,0.00012907872,0.00022015742,0.0034148404,0.0015555358,0.0062043383],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995778,0.000053026248,0.000020730295,0.00017865785,0.000110252484,0.000059551043],"domain_scores_gemma":[0.9995584,0.00017091758,0.000045999474,0.00008876767,0.00008666995,0.000049340353],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007861003,0.0014581819,0.0013364167,0.0011243022,0.0004362019,0.0010439169,0.0023543998,0.0017677433,0.0031770803],"category_scores_gemma":[0.002779681,0.00081595976,0.001237954,0.0008320565,0.0007428487,0.0019179628,0.0014878155,0.0023042327,0.0015064942],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047621902,0.0001836111,0.0018454603,0.00026110356,0.00019814764,0.00023889374,0.00018127654,0.45606184,0.04039235,0.020083044,0.013061377,0.46701664],"study_design_scores_gemma":[0.000009948036,0.000028955803,0.00013932322,0.00000886714,0.000008156191,0.00004739917,0.000010577416,0.9887057,0.005271404,0.0036820902,0.0020783981,0.000009310994],"about_ca_topic_score_codex":0.012225337,"about_ca_topic_score_gemma":0.016904758,"teacher_disagreement_score":0.012225337,"about_ca_system_score_codex":0.0012358858,"about_ca_system_score_gemma":0.0013048385,"threshold_uncertainty_score":0.024308324},"labels":[],"label_agreement":null},{"id":"W4413155320","doi":"10.1109/cvpr52734.2025.01341","title":"BACON: Improving Clarity of Image Captions via Bag-of-Concept Graphs","year":2025,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"CLARITY; Computer science; Artificial intelligence; Image (mathematics); Natural language processing; Information retrieval; Computer vision; Chemistry","score_opus":0.0067285771908953814,"score_gpt":0.268057067877984,"score_spread":0.26132849068708863,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413155320","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019508965,0.003614696,0.81892145,0.0011875075,0.0015384993,0.00095822033,0.013253538,0.12622397,0.014793142],"genre_scores_gemma":[0.13823126,0.001798726,0.7985816,0.0017685329,0.00035886437,0.0011588018,0.036463287,0.010552249,0.01108672],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9989429,0.0003187259,0.00005467702,0.00038453055,0.00020592667,0.00009323806],"domain_scores_gemma":[0.99735177,0.0012554063,0.00016937537,0.0005199432,0.0006004312,0.00010307081],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013775115,0.0042660665,0.0009235884,0.0032962223,0.0009363452,0.0024765227,0.0029757314,0.0028488203,0.01389033],"category_scores_gemma":[0.008204624,0.001017983,0.002741213,0.0019935975,0.0011106805,0.0044776164,0.0027221644,0.0038161702,0.007209098],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00073001336,0.00022678838,0.001632973,0.0025109905,0.00028557976,0.00089927163,0.0012176449,0.12806752,0.03360787,0.021345804,0.26883623,0.54063934],"study_design_scores_gemma":[0.00010666156,0.0001782117,0.00082724885,0.000325077,0.00012571517,0.00046748875,0.00028791427,0.83193225,0.027696785,0.028080361,0.10983583,0.00013644877],"about_ca_topic_score_codex":0.010032875,"about_ca_topic_score_gemma":0.014963967,"teacher_disagreement_score":0.01389033,"about_ca_system_score_codex":0.0018053511,"about_ca_system_score_gemma":0.0014370101,"threshold_uncertainty_score":0.04646778},"labels":[],"label_agreement":null},{"id":"W4413769926","doi":"10.1097/wno.0000000000002393","title":"Artificial Intelligence Diagnosis of Ocular Motility Disorders From Clinical Videos","year":2025,"lang":"en","type":"article","venue":"Journal of Neuro-Ophthalmology","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Western Hospital","funders":"","keywords":"Computer science; Medicine; Artificial intelligence","score_opus":0.05013311516629099,"score_gpt":0.3844152274446197,"score_spread":0.3342821122783287,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413769926","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.87972414,0.0040466897,0.0854579,0.0010200837,0.00018899239,0.0011257277,0.011270227,0.0035440216,0.013622145],"genre_scores_gemma":[0.9077698,0.0015600476,0.08105392,0.00019408387,0.00014058269,0.00029565452,0.007168521,0.000069273054,0.0017481372],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99958545,0.000101826365,0.000055843513,0.00009209946,0.00012907006,0.000035687633],"domain_scores_gemma":[0.998495,0.0006705904,0.0002300138,0.00008649877,0.00043831198,0.000079734134],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00085238676,0.0004785994,0.00023634867,0.002128017,0.00012279958,0.00061490457,0.0003912193,0.00030365642,0.0019119086],"category_scores_gemma":[0.0050300052,0.00009600953,0.00017595953,0.000568175,0.00014213337,0.00038522834,0.0005511795,0.00018930649,0.00066457427],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013528022,0.00030379236,0.19996163,0.001382522,0.00017450762,0.0020961782,0.0005628003,0.007908261,0.053080816,0.000939256,0.022147702,0.71008974],"study_design_scores_gemma":[0.00020844692,0.0012608159,0.44502494,0.0008374387,0.00032895053,0.010491495,0.0014869685,0.40168253,0.10321854,0.0039444733,0.031378105,0.00013728348],"about_ca_topic_score_codex":0.001956205,"about_ca_topic_score_gemma":0.0030388564,"teacher_disagreement_score":0.002128017,"about_ca_system_score_codex":0.00038529647,"about_ca_system_score_gemma":0.00044518593,"threshold_uncertainty_score":0.0063959956},"labels":[],"label_agreement":null},{"id":"W4413944815","doi":"10.1109/icra55743.2025.11128004","title":"OLiVia-Nav: An Online Lifelong Vision Language Approach for Mobile Robot Social Navigation","year":2025,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Mobile robot; Mobile robot navigation; Human–computer interaction; Robot; Computer vision; Social robot; Artificial intelligence; Multimedia; Robot control","score_opus":0.018775548245875733,"score_gpt":0.35996635047231645,"score_spread":0.3411908022264407,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413944815","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017408712,0.0002463966,0.97436124,0.0001876279,0.00008909303,0.000064382795,0.00012519435,0.0050094095,0.0025078696],"genre_scores_gemma":[0.47867277,0.0002261157,0.508054,0.00057928945,0.00006856808,0.0002494135,0.0008362047,0.0006188065,0.010694741],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997044,0.00006816051,0.0000121007415,0.0001093711,0.0000644383,0.000041429375],"domain_scores_gemma":[0.9996382,0.000108045926,0.00003855911,0.00008126426,0.000093794166,0.000040079532],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005174897,0.00073263847,0.00047194416,0.00032657027,0.00036029762,0.00056880683,0.0020069582,0.00090096076,0.0025851154],"category_scores_gemma":[0.0016439056,0.00030377024,0.00063139776,0.00022434321,0.0005451761,0.0017433364,0.0018397446,0.0014471697,0.0010675925],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025455785,0.0003570057,0.0017931551,0.00018849476,0.0001041827,0.00023832926,0.00048050945,0.24387985,0.03329547,0.019730933,0.0109823905,0.6886951],"study_design_scores_gemma":[0.000012619452,0.0000927851,0.00016338192,0.000009079729,0.000011289824,0.000053185235,0.000046246754,0.98400694,0.00507936,0.0067074513,0.003800336,0.000017290646],"about_ca_topic_score_codex":0.0064313193,"about_ca_topic_score_gemma":0.012033818,"teacher_disagreement_score":0.0064313193,"about_ca_system_score_codex":0.00062234857,"about_ca_system_score_gemma":0.0011071707,"threshold_uncertainty_score":0.012787759},"labels":[],"label_agreement":null},{"id":"W4413986970","doi":"10.14778/3748191.3748195","title":"Déjà Vu: Efficient Video-Language Query Engine with Learning-Based Inter-Frame Computation Reuse","year":2025,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Computer science; Reuse; Déjà vu; Frame (networking); Computation; Artificial intelligence; Natural language processing; Programming language; Computer network; Engineering; Psychology","score_opus":0.0037931050791894048,"score_gpt":0.24165399323101416,"score_spread":0.23786088815182477,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413986970","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016738588,0.0006863781,0.9098299,0.00026856852,0.00017663414,0.00020975144,0.000969156,0.068568826,0.0025521358],"genre_scores_gemma":[0.2655531,0.00048520198,0.71235996,0.00061436347,0.00011419075,0.00044830196,0.006289948,0.0038264822,0.010308537],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991026,0.000108472224,0.000061059225,0.00029672464,0.0003348884,0.000096172764],"domain_scores_gemma":[0.9991986,0.00026715797,0.000054332053,0.00022245738,0.00019572704,0.000061798935],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00096841686,0.0014646769,0.00088507467,0.0007565432,0.00040791652,0.0016018403,0.0036867198,0.0010507066,0.005818575],"category_scores_gemma":[0.0044962894,0.0005685488,0.0008574761,0.00076104584,0.0006027001,0.0036540786,0.00277851,0.0015904299,0.003189101],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012546633,0.0004435887,0.0024144184,0.00047196314,0.00016468161,0.0004448416,0.00036402515,0.09586532,0.06782362,0.025041835,0.07743357,0.72827756],"study_design_scores_gemma":[0.00007018345,0.00011628122,0.0001778118,0.00000984751,0.000016736265,0.00011064248,0.000053829943,0.95374894,0.028600836,0.0070222253,0.010042507,0.00003021332],"about_ca_topic_score_codex":0.009119345,"about_ca_topic_score_gemma":0.013079966,"teacher_disagreement_score":0.009119345,"about_ca_system_score_codex":0.0010095306,"about_ca_system_score_gemma":0.0014366672,"threshold_uncertainty_score":0.019465089},"labels":[],"label_agreement":null},{"id":"W4414317082","doi":"10.20944/preprints202509.1607.v1","title":"Capturing Narrative Semantics from Captions for Relational Scene Abstraction","year":2025,"lang":"en","type":"preprint","venue":"Preprints.org","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Interpretability; Scene graph; Closed captioning; Semantics (computer science); Graph; Scalability; Natural language; Exploit; Abstraction","score_opus":0.14039488549821855,"score_gpt":0.38429230633133954,"score_spread":0.243897420833121,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414317082","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027491156,0.00050462486,0.94312996,0.00036389029,0.00009559905,0.00033649677,0.003019937,0.019867811,0.0051904377],"genre_scores_gemma":[0.35730413,0.00056298135,0.6198247,0.00031421622,0.0000773837,0.00032832898,0.0146501595,0.002284465,0.0046536284],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994635,0.0001422187,0.000023093984,0.00023093025,0.00010235319,0.000037858004],"domain_scores_gemma":[0.9989643,0.0004113752,0.00010921136,0.00032145384,0.00013918999,0.000054392145],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00055322587,0.0018222834,0.0004940072,0.0014100523,0.0005061019,0.0016321323,0.0019298511,0.0010713163,0.007365675],"category_scores_gemma":[0.004015193,0.0007900497,0.0016954918,0.00073675497,0.0008783362,0.0033313644,0.0018954661,0.0019512716,0.002320816],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040237783,0.00027209063,0.00420371,0.0012982538,0.00020395056,0.0011272811,0.00212008,0.23850173,0.058898825,0.07260773,0.043661322,0.5767028],"study_design_scores_gemma":[0.000022933713,0.000073803945,0.0009005655,0.00006909607,0.000051586096,0.0002541487,0.0002727286,0.9099052,0.0165245,0.050342534,0.021548728,0.000034167933],"about_ca_topic_score_codex":0.0037995416,"about_ca_topic_score_gemma":0.009109168,"teacher_disagreement_score":0.007365675,"about_ca_system_score_codex":0.0009372485,"about_ca_system_score_gemma":0.00062238,"threshold_uncertainty_score":0.02464062},"labels":[],"label_agreement":null},{"id":"W4414857280","doi":"10.48550/arxiv.2505.24434","title":"Graph Flow Matching: Enhancing Image Generation with Neighbor-Aware Flow Fields","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Pointwise; Flow (mathematics); Vector field; Matching (statistics); Graph; Flow velocity; Scalability; Artificial neural network; Pattern recognition (psychology)","score_opus":0.022610567812840356,"score_gpt":0.27731955546846776,"score_spread":0.2547089876556274,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414857280","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03356373,0.0004899258,0.95511097,0.00037579142,0.00018637456,0.00012128167,0.0003072557,0.0068087443,0.0030358816],"genre_scores_gemma":[0.4625063,0.000327089,0.5280272,0.0005666403,0.00011513995,0.00014783486,0.0014712396,0.00095055345,0.005888035],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995454,0.000073860014,0.000018627105,0.0001669359,0.0001420926,0.000052997377],"domain_scores_gemma":[0.99933165,0.00025293443,0.000063392414,0.00015396226,0.00014438195,0.000053616044],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010822713,0.0010652838,0.0006616974,0.0012195099,0.0003641616,0.0009611126,0.001995529,0.0013849227,0.0033557985],"category_scores_gemma":[0.0035921617,0.00041922825,0.00078486116,0.0008511372,0.0006336706,0.0018125647,0.0014950986,0.0015666628,0.0012004423],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002962906,0.00026085175,0.0021207936,0.00014983628,0.00007906878,0.0001306261,0.00012387308,0.34888294,0.026476117,0.012727223,0.01191653,0.5968358],"study_design_scores_gemma":[0.000019587658,0.000034228724,0.00020691282,0.0000074369614,0.000008886069,0.000031702708,0.000009645405,0.98611605,0.006738124,0.0055413875,0.0012784231,0.000007663771],"about_ca_topic_score_codex":0.008486885,"about_ca_topic_score_gemma":0.010557675,"teacher_disagreement_score":0.008486885,"about_ca_system_score_codex":0.0010415055,"about_ca_system_score_gemma":0.0010121096,"threshold_uncertainty_score":0.016874969},"labels":[],"label_agreement":null},{"id":"W4415184565","doi":"10.36227/techrxiv.176049757.72015578/v1","title":"Lightweight Adaptation of Large Language and Vision Models in Robotics","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Robotics; Adaptation (eye); Machine vision; Robot; Foundation (evidence); Field (mathematics)","score_opus":0.015941705709779763,"score_gpt":0.3216170326830381,"score_spread":0.30567532697325833,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415184565","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011292224,0.0032672249,0.97982556,0.00032691105,0.0000944657,0.000045703287,0.00011738073,0.0024455066,0.0025850204],"genre_scores_gemma":[0.59609216,0.005632016,0.388186,0.0008616935,0.00020861348,0.00044760745,0.0009109672,0.0011584709,0.0065024253],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99966073,0.000079231235,0.000025008894,0.000090826754,0.00008832142,0.000055828976],"domain_scores_gemma":[0.99956256,0.00023414691,0.000035351368,0.000076217664,0.000069957605,0.00002177592],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007193341,0.0013791063,0.0010365888,0.00039673096,0.0003367659,0.0011002744,0.0016414633,0.0012967403,0.0027794312],"category_scores_gemma":[0.002792816,0.0005548936,0.0009565577,0.00055904716,0.0007915847,0.001697517,0.0015578972,0.0022378785,0.0012570178],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005620417,0.0000612622,0.00050428085,0.0003826983,0.0000872884,0.00010620571,0.00011269252,0.77250254,0.009598543,0.017564908,0.0037521715,0.19527125],"study_design_scores_gemma":[0.000005239411,0.00003105849,0.00010771751,0.000026310478,0.000011987758,0.00003670952,0.00001462177,0.98693347,0.0015743987,0.009161714,0.0020848305,0.000011875505],"about_ca_topic_score_codex":0.004391909,"about_ca_topic_score_gemma":0.0051130718,"teacher_disagreement_score":0.004391909,"about_ca_system_score_codex":0.00070302916,"about_ca_system_score_gemma":0.0011251363,"threshold_uncertainty_score":0.009298086},"labels":[],"label_agreement":null},{"id":"W4415214274","doi":"10.1016/j.neucom.2025.131217","title":"Sycophancy in vision-language models: A systematic analysis and an inference-time mitigation framework","year":2025,"lang":"en","type":"article","venue":"Neurocomputing","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Key Research and Development Program of China; Youth Innovation Promotion Association of the Chinese Academy of Sciences; Chinese Academy of Sciences; National Natural Science Foundation of China; Canadian Anesthesiologists' Society","keywords":"Sensitivity (control systems); Inference; Filter (signal processing); Trustworthiness; Decoding methods; Polarity (international relations); Path (computing); Mechanism (biology)","score_opus":0.0071095148556247585,"score_gpt":0.32275936569275615,"score_spread":0.3156498508371314,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415214274","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009165446,0.0013065264,0.9861691,0.001252247,0.00008694299,0.000070525246,0.00010242415,0.00028879193,0.0015580151],"genre_scores_gemma":[0.687545,0.004856156,0.28865084,0.0014559025,0.0012507016,0.0005889939,0.00078228663,0.00084931235,0.014020773],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9944602,0.00273133,0.00027067878,0.0010886915,0.0010514947,0.00039753408],"domain_scores_gemma":[0.9586277,0.03494882,0.0016455614,0.002561527,0.0018512451,0.00036505258],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010885331,0.0019461067,0.0021508653,0.0017451787,0.0014536711,0.0026277807,0.0033005192,0.0027182603,0.0038207828],"category_scores_gemma":[0.045238245,0.0017045018,0.0021719104,0.0012034007,0.0033459712,0.0071051694,0.0054163216,0.0066205086,0.00065733155],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004460052,0.00023334997,0.0044732676,0.00085009314,0.00052827393,0.0010282602,0.0006310016,0.31220645,0.006228454,0.52910674,0.007836946,0.13643117],"study_design_scores_gemma":[0.000016214326,0.00007386543,0.00047399558,0.00006848918,0.00009862291,0.00015512218,0.0000494948,0.8528903,0.0020563703,0.14252472,0.0015646083,0.000028219203],"about_ca_topic_score_codex":0.0061689597,"about_ca_topic_score_gemma":0.0063261245,"teacher_disagreement_score":0.010885331,"about_ca_system_score_codex":0.0019813115,"about_ca_system_score_gemma":0.004188988,"threshold_uncertainty_score":0.057567835},"labels":[],"label_agreement":null},{"id":"W4415306945","doi":"10.1109/iccv51701.2025.00863","title":"Perspective-Aware Reasoning in Vision-Language Models via Mental Imagery Simulation","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Perspective (graphical); Visual reasoning; Focus (optics); Construct (python library); Bridge (graph theory); Mental image; Spatial intelligence; Orientation (vector space); Object (grammar)","score_opus":0.013292323184660523,"score_gpt":0.34967133245220955,"score_spread":0.33637900926754905,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415306945","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016971014,0.00011908302,0.9783321,0.000283565,0.000030467796,0.000051604202,0.0001391615,0.00083324756,0.0032396622],"genre_scores_gemma":[0.59039944,0.00022529792,0.4063222,0.00018539412,0.000038862425,0.00022873793,0.00041519108,0.00021044421,0.0019744167],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994336,0.0002029418,0.000028204133,0.00012126868,0.00015074553,0.000063220104],"domain_scores_gemma":[0.99884355,0.0006305986,0.00011545432,0.00021403054,0.00010482713,0.000091526075],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007570635,0.00077785284,0.0006714439,0.00045735933,0.00047244356,0.0019846465,0.0021230935,0.00115237,0.0034022632],"category_scores_gemma":[0.003918524,0.00046990992,0.0018472837,0.00032753308,0.0014258146,0.002337083,0.0023277346,0.0019479239,0.00056143565],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013694489,0.00007822629,0.0006889152,0.00013218797,0.000066446795,0.00020278865,0.00042769697,0.832086,0.005030672,0.13031055,0.001350541,0.029489022],"study_design_scores_gemma":[0.00001376764,0.000015216676,0.00003654167,0.0000070797187,0.000007841409,0.000019749088,0.00002102825,0.9581515,0.00086500327,0.04002745,0.0008275168,0.0000073183146],"about_ca_topic_score_codex":0.0065565268,"about_ca_topic_score_gemma":0.0077778455,"teacher_disagreement_score":0.0065565268,"about_ca_system_score_codex":0.001137604,"about_ca_system_score_gemma":0.0013317195,"threshold_uncertainty_score":0.013036728},"labels":[],"label_agreement":null},{"id":"W4415535740","doi":"10.1016/j.array.2025.100538","title":"Mamba-caption: Long-range sequence modelling for efficient and accurate image captioning","year":2025,"lang":"en","type":"article","venue":"Array","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Moncton","funders":"University of Johannesburg","keywords":"Closed captioning; Security token; Convolutional neural network; Trigram; Embedding; Clipping (morphology); Decoding methods; Image (mathematics); Pattern recognition (psychology)","score_opus":0.029847909417003684,"score_gpt":0.31212076700627617,"score_spread":0.2822728575892725,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415535740","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005929156,0.00073634286,0.97700286,0.00040764525,0.00024421516,0.00012264344,0.00069064874,0.011081581,0.003784905],"genre_scores_gemma":[0.26004833,0.0012301096,0.7131578,0.000833732,0.00023293351,0.00056028913,0.0043738666,0.002865275,0.016697658],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99946445,0.00015618894,0.000033112072,0.00017421959,0.00011909705,0.000052920866],"domain_scores_gemma":[0.9989003,0.0005309767,0.00006376069,0.00022485308,0.00022881104,0.000051318344],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011420621,0.0014036596,0.000800936,0.0006624313,0.00045083143,0.0017960755,0.002400431,0.0017782823,0.008609783],"category_scores_gemma":[0.0046833116,0.00062905496,0.0011570492,0.00073849614,0.0008892574,0.0030949647,0.0016468663,0.0025406664,0.0050323806],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00050592294,0.00014667229,0.0008150501,0.00081909273,0.000115168106,0.0004998349,0.0004830781,0.42626804,0.05330022,0.048083432,0.038859084,0.43010446],"study_design_scores_gemma":[0.000013996056,0.000054925637,0.00008416074,0.000025004836,0.000013516628,0.00010420958,0.000027033013,0.9608069,0.014920728,0.013618726,0.010312179,0.000018662218],"about_ca_topic_score_codex":0.00454639,"about_ca_topic_score_gemma":0.008080633,"teacher_disagreement_score":0.008609783,"about_ca_system_score_codex":0.0010972291,"about_ca_system_score_gemma":0.0011691244,"threshold_uncertainty_score":0.028802574},"labels":[],"label_agreement":null},{"id":"W4415540597","doi":"10.1145/3746027.3755141","title":"Twin Co-Adaptive Dialogue for Progressive Image Generation","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Workflow; Process (computing); Image (mathematics); Quality (philosophy); Iterative and incremental development; Base (topology); Series (stratigraphy); Conjunction (astronomy)","score_opus":0.0294783400563756,"score_gpt":0.3478488747859853,"score_spread":0.3183705347296097,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415540597","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018573163,0.00021072698,0.9725458,0.00010319082,0.000055103374,0.00020302247,0.000034288547,0.0046775155,0.003597201],"genre_scores_gemma":[0.39902508,0.00013126817,0.5922265,0.0002108522,0.00006334096,0.00045025817,0.00018491525,0.0011716224,0.0065361126],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99805874,0.0008657051,0.00007590852,0.00048911985,0.00038592823,0.0001245798],"domain_scores_gemma":[0.9968292,0.0018827133,0.00015519052,0.0005098085,0.00038588978,0.00023721615],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023484612,0.0011213858,0.0005962407,0.00062904175,0.00080463546,0.0013286537,0.0022029162,0.0013642008,0.008126054],"category_scores_gemma":[0.008622205,0.00050623674,0.00062476343,0.00028683685,0.0012251461,0.0019839646,0.004461574,0.0012367535,0.0016854737],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018645726,0.0008863817,0.003067495,0.00065378635,0.00015236497,0.0013430546,0.008835658,0.09935361,0.19353884,0.04332261,0.013814765,0.6331669],"study_design_scores_gemma":[0.00016033764,0.00037792293,0.00093519076,0.000051012543,0.00005048901,0.0006729281,0.0004594239,0.89734375,0.04864032,0.026142383,0.025043765,0.00012237168],"about_ca_topic_score_codex":0.0011134164,"about_ca_topic_score_gemma":0.0014227051,"teacher_disagreement_score":0.008126054,"about_ca_system_score_codex":0.00039006412,"about_ca_system_score_gemma":0.0005893931,"threshold_uncertainty_score":0.027184367},"labels":[],"label_agreement":null},{"id":"W4415623941","doi":"10.21203/rs.3.rs-7752202/v1","title":"Towards Multimodal Retrieval-Augmented Generation for Medical Visual Question Answering","year":2025,"lang":"","type":"preprint","venue":"Research Square","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Task (project management); Question answering; Trustworthiness; Reliability (semiconductor); Natural language generation; Natural language; Health care; Medical information","score_opus":0.06851537538957315,"score_gpt":0.4775223534019001,"score_spread":0.40900697801232694,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415623941","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018271688,0.0012796249,0.95110714,0.0011699869,0.00044931975,0.00040852546,0.0021467067,0.01906922,0.006097662],"genre_scores_gemma":[0.25976843,0.0006863634,0.7141058,0.0010591856,0.00041124833,0.0005466689,0.008466264,0.00153726,0.013418784],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985592,0.0006170508,0.00007337537,0.00037043128,0.00022526235,0.00015478881],"domain_scores_gemma":[0.99808866,0.0009177174,0.00006261999,0.00042635758,0.00041738083,0.00008732539],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016573344,0.0013902467,0.0013938891,0.0015317824,0.0005914569,0.00215395,0.0021511489,0.0029917606,0.02076743],"category_scores_gemma":[0.005930577,0.0006150429,0.0019819434,0.0008575656,0.00079111266,0.0020916397,0.0034150633,0.0017940523,0.008127565],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001155405,0.000494817,0.0009803887,0.00076254114,0.00016968475,0.00055531255,0.0006130358,0.031135662,0.07929469,0.016503727,0.054670386,0.8136644],"study_design_scores_gemma":[0.00014913868,0.0002638573,0.00083802955,0.000089877954,0.00013093239,0.0003989309,0.00025403773,0.8955763,0.040230684,0.038755093,0.023248537,0.00006457019],"about_ca_topic_score_codex":0.0044610426,"about_ca_topic_score_gemma":0.004824744,"teacher_disagreement_score":0.02076743,"about_ca_system_score_codex":0.00081840873,"about_ca_system_score_gemma":0.0009515147,"threshold_uncertainty_score":0.06947392},"labels":[],"label_agreement":null},{"id":"W4415822394","doi":"10.1109/ro-man63969.2025.11217693","title":"Transparent Social Navigation for Autonomous Mobile Robots Via Vision-Language Models","year":2025,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University","keywords":"Robot; Transparency (behavior); Confusion; Mobile robot; Preference; Service (business); Natural language; Social robot","score_opus":0.016128799605941622,"score_gpt":0.3404957217373021,"score_spread":0.3243669221313605,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415822394","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09944291,0.00012423155,0.8957038,0.0004315437,0.000025720257,0.00010684792,0.000116324496,0.0026408357,0.0014078681],"genre_scores_gemma":[0.872714,0.00008834863,0.12562059,0.000080284,0.000012844545,0.00012436206,0.00017184729,0.00009212229,0.0010956409],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9992686,0.0003835588,0.00003349467,0.00012941757,0.0001359342,0.00004904044],"domain_scores_gemma":[0.998279,0.0009851943,0.00025992072,0.00020606807,0.00020783348,0.000062128],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010868536,0.00068361795,0.0002707186,0.00040397397,0.0004011979,0.0011474071,0.0005968758,0.0006897207,0.0016324741],"category_scores_gemma":[0.006044253,0.00026499195,0.0007076572,0.00021499164,0.0005988728,0.0021935522,0.0016890302,0.0008696185,0.00039332637],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011498823,0.0006726242,0.010642613,0.00051033625,0.00026597833,0.0010191815,0.009295781,0.35045147,0.09365301,0.042800475,0.005687263,0.4838514],"study_design_scores_gemma":[0.000027879052,0.00016398843,0.0014570244,0.000029652909,0.000038635535,0.00010727979,0.00044449492,0.96790755,0.009279021,0.017907932,0.0025932984,0.000043222808],"about_ca_topic_score_codex":0.0055700284,"about_ca_topic_score_gemma":0.0062697357,"teacher_disagreement_score":0.0055700284,"about_ca_system_score_codex":0.00051550835,"about_ca_system_score_gemma":0.0007263833,"threshold_uncertainty_score":0.011075199},"labels":[],"label_agreement":null},{"id":"W4415848069","doi":"10.1051/wujns/2025305405","title":"Navigating with Spatial Intelligence: A Survey of Scene Graph-Based Object Goal Navigation","year":2025,"lang":"","type":"article","venue":"Wuhan University Journal of Natural Sciences","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Science and Technology Program of Hubei Province","keywords":"Focus (optics); Graph; Object (grammar); Field (mathematics); Robot; Intelligent agent; Scene graph; Generalization","score_opus":0.01439905649340406,"score_gpt":0.2964890693527256,"score_spread":0.28209001285932156,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415848069","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017547583,0.120379984,0.81288743,0.0019069633,0.0005141095,0.0001849949,0.0006518907,0.0018259856,0.04410104],"genre_scores_gemma":[0.32985225,0.21006191,0.44092724,0.0010351171,0.0005178884,0.0003566648,0.0031835893,0.0005730477,0.013492297],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99973553,0.000048088834,0.00001906784,0.000091606846,0.00008289038,0.000022774864],"domain_scores_gemma":[0.9997261,0.00010530771,0.000026648564,0.000038422713,0.00007790894,0.000025530391],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002892171,0.00093642605,0.00072507065,0.0015714533,0.00052639627,0.0012826528,0.0014171273,0.00075074926,0.0023104027],"category_scores_gemma":[0.0008788681,0.0003410599,0.0009942073,0.0028759665,0.0006454902,0.0028679862,0.001197063,0.00068531564,0.0007254484],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006784867,0.00013782263,0.0035185064,0.0027743438,0.00016208252,0.0002351751,0.0007437867,0.065699,0.0028026388,0.13032639,0.018225957,0.77530646],"study_design_scores_gemma":[0.00003766435,0.0002800599,0.0058253,0.0010957137,0.00037893222,0.0009828267,0.001252923,0.38345918,0.0041640075,0.21166265,0.39065772,0.00020314722],"about_ca_topic_score_codex":0.015356903,"about_ca_topic_score_gemma":0.012409711,"teacher_disagreement_score":0.015356903,"about_ca_system_score_codex":0.00096472685,"about_ca_system_score_gemma":0.0013411676,"threshold_uncertainty_score":0.030535042},"labels":[],"label_agreement":null},{"id":"W4415871599","doi":"10.36227/techrxiv.174612962.26131807/v4","title":"LLM-Based Human-Agent Collaboration and Interaction Systems: A Survey","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Face (sociological concept); Reliability (semiconductor); Field (mathematics); Identification (biology); Action (physics)","score_opus":0.028474859388193614,"score_gpt":0.3582016073138506,"score_spread":0.329726747925657,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415871599","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01657782,0.20571797,0.7192006,0.0041501806,0.00057697995,0.0006981784,0.00049165403,0.004480527,0.048106045],"genre_scores_gemma":[0.3389425,0.14145555,0.49605766,0.0018110923,0.0009640187,0.0013401879,0.0019983219,0.00076173217,0.016668916],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99674714,0.0013203814,0.00030324518,0.0005952177,0.00089224905,0.00014184773],"domain_scores_gemma":[0.9968033,0.0019178734,0.00025722542,0.0004272953,0.0004139767,0.0001802825],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030879695,0.0009822419,0.0012437713,0.0018353567,0.0011579766,0.0035924227,0.0028869363,0.0020979485,0.0065077986],"category_scores_gemma":[0.0066584647,0.00068972557,0.0009419817,0.0023148276,0.0012408497,0.004552845,0.0036928477,0.0011165363,0.0022482974],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019088024,0.00027668563,0.0031164268,0.0061804517,0.00021825917,0.00030381817,0.0019003381,0.024681175,0.003918254,0.095165916,0.017509488,0.8465383],"study_design_scores_gemma":[0.00006502629,0.0005462125,0.0031487455,0.0021640123,0.00019419895,0.0015343074,0.0020745296,0.21461524,0.0061446857,0.14341706,0.62590736,0.00018871312],"about_ca_topic_score_codex":0.002588467,"about_ca_topic_score_gemma":0.001909307,"teacher_disagreement_score":0.0065077986,"about_ca_system_score_codex":0.0014617209,"about_ca_system_score_gemma":0.0019036785,"threshold_uncertainty_score":0.021770716},"labels":[],"label_agreement":null},{"id":"W4415955467","doi":"10.1109/iccv51701.2025.00074","title":"MultiVerse: A Multi-Turn Conversation Benchmark for Evaluating Large Vision and Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Ministry of Science and ICT, South Korea","keywords":"Conversation; Benchmark (surveying); Set (abstract data type); Perception; Context (archaeology); Key (lock)","score_opus":0.02903482643643528,"score_gpt":0.38900244767120323,"score_spread":0.35996762123476794,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415955467","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.49774477,0.012778656,0.35193297,0.0021517053,0.0024178342,0.003268063,0.03577909,0.058530804,0.035396207],"genre_scores_gemma":[0.6690353,0.0009842737,0.24585131,0.0010506122,0.00031644633,0.0025151707,0.06993712,0.0027817907,0.007527985],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9908229,0.004913836,0.0006803318,0.0019412016,0.0011891477,0.00045253654],"domain_scores_gemma":[0.98839635,0.007648197,0.0005237299,0.0012888825,0.0013252092,0.00081756985],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008396982,0.0048037283,0.0014537458,0.0029904637,0.0015007121,0.0028926807,0.0034654778,0.002968597,0.0055230176],"category_scores_gemma":[0.028169928,0.0006302134,0.0020319433,0.0012608687,0.0010664079,0.0041350676,0.005783799,0.0034053803,0.0036637974],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0051905,0.0032767553,0.030796304,0.0049664904,0.0021624267,0.00091201323,0.0027547218,0.15954256,0.026737565,0.006909142,0.12248558,0.63426596],"study_design_scores_gemma":[0.0005071021,0.0031044325,0.014564872,0.00038633335,0.00033960328,0.0008580226,0.002201718,0.8968257,0.028411526,0.013090321,0.039382983,0.0003273927],"about_ca_topic_score_codex":0.010696856,"about_ca_topic_score_gemma":0.014952177,"teacher_disagreement_score":0.010696856,"about_ca_system_score_codex":0.0016192603,"about_ca_system_score_gemma":0.0018833391,"threshold_uncertainty_score":0.044408023},"labels":[],"label_agreement":null},{"id":"W4416034471","doi":"10.18653/v1/2025.findings-emnlp.1340","title":"QEVA: A Reference-Free Evaluation Metric for Narrative Video Summarization with Multimodal Question Answering","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; Chung-Ang University","keywords":"Automatic summarization; Question answering; Metric (unit); Narrative; Key (lock)","score_opus":0.024177218021763723,"score_gpt":0.3427726956916791,"score_spread":0.3185954776699154,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416034471","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21942718,0.016828155,0.6675365,0.0011689558,0.0011120159,0.0021611035,0.041141924,0.032762296,0.017861875],"genre_scores_gemma":[0.5709931,0.0015319765,0.35005224,0.00033552403,0.00030220716,0.0017646922,0.06969469,0.0012553423,0.004070182],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9904937,0.0043879743,0.0011666855,0.0014528721,0.0022153454,0.0002834532],"domain_scores_gemma":[0.97488934,0.014212604,0.0020152384,0.0023922331,0.005846661,0.00064397627],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0073238565,0.0020491136,0.0012189709,0.0070193824,0.0008793766,0.0023027007,0.0017378109,0.0016958598,0.004100266],"category_scores_gemma":[0.051805142,0.0002588345,0.0008034107,0.0036483903,0.0005922579,0.0042450447,0.0027008462,0.0010818127,0.00211162],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022296682,0.00053934974,0.020235611,0.004801234,0.0009528255,0.00035234797,0.0015749322,0.033506583,0.035104737,0.0064786784,0.069259614,0.82496446],"study_design_scores_gemma":[0.00043266482,0.006107138,0.071019515,0.0010290259,0.00082382193,0.0016961498,0.0030274475,0.7043587,0.08299372,0.026883665,0.10102236,0.0006057361],"about_ca_topic_score_codex":0.004423648,"about_ca_topic_score_gemma":0.006069425,"teacher_disagreement_score":0.0073238565,"about_ca_system_score_codex":0.0012663768,"about_ca_system_score_gemma":0.0011736009,"threshold_uncertainty_score":0.038732767},"labels":[],"label_agreement":null},{"id":"W4416035046","doi":"10.18653/v1/2025.findings-emnlp.387","title":"AdaptMerge: Inference Time Adaptive Visual and Language-Guided Token Merging for Efficient Large Multimodal Models","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"Natural Sciences and Engineering Research Council of Canada; DeepMind","keywords":"Inference; Security token; Feature (linguistics); Visualization; Pattern recognition (psychology)","score_opus":0.0188835584601864,"score_gpt":0.3386773881967409,"score_spread":0.3197938297365545,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416035046","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015543384,0.0007117488,0.95990306,0.00038775132,0.00016024837,0.00012488275,0.00054075907,0.020685434,0.0019427638],"genre_scores_gemma":[0.31781307,0.00038714276,0.66699266,0.00079509406,0.0001308164,0.00042721524,0.0028022751,0.0028316488,0.00782004],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.999331,0.00017004313,0.000033087133,0.00023272305,0.00013840107,0.00009470745],"domain_scores_gemma":[0.9991062,0.00043258385,0.00004469894,0.00022936142,0.00012169287,0.00006544884],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012706395,0.0017553972,0.0014141835,0.0008768088,0.0007866222,0.001418591,0.0038332462,0.0014696816,0.008256493],"category_scores_gemma":[0.0047859373,0.0008405385,0.0014512222,0.00088817975,0.00096249365,0.003378362,0.0032002039,0.0026942322,0.0030583893],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00075508363,0.00022008068,0.0013730895,0.00032524916,0.00024041382,0.00029397296,0.00028259485,0.33961204,0.016379826,0.016086398,0.026005857,0.59842545],"study_design_scores_gemma":[0.000034687935,0.000044110333,0.00011731589,0.000010501752,0.000022028102,0.000051246876,0.000042858483,0.97777903,0.005444339,0.013444306,0.0029922926,0.000017260665],"about_ca_topic_score_codex":0.011792471,"about_ca_topic_score_gemma":0.023997251,"teacher_disagreement_score":0.011792471,"about_ca_system_score_codex":0.0016053842,"about_ca_system_score_gemma":0.0021749642,"threshold_uncertainty_score":0.027620733},"labels":[],"label_agreement":null},{"id":"W4416035530","doi":"10.18653/v1/2025.emnlp-main.1379","title":"CAVE : Detecting and Explaining Commonsense Anomalies in Visual Environments","year":2025,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Anomaly detection; Cave; Perception; Visual reasoning; Commonsense reasoning; Visualization; Cognition; Resource (disambiguation)","score_opus":0.007723986487770126,"score_gpt":0.27797688700207784,"score_spread":0.2702529005143077,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416035530","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.070725165,0.0023262454,0.06009723,0.002078975,0.0008261803,0.0015771092,0.66889703,0.17489508,0.018577047],"genre_scores_gemma":[0.07233521,0.00033159566,0.07550322,0.0005392566,0.00007764499,0.00086019014,0.84207976,0.0027790617,0.0054940316],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982578,0.00038329363,0.00009786386,0.00068719714,0.0004105855,0.00016337726],"domain_scores_gemma":[0.9962942,0.001658704,0.0002419417,0.0011116874,0.0004620346,0.00023144302],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012781023,0.0032773784,0.000948921,0.0027118216,0.00091083377,0.0023712057,0.0045019086,0.0029084987,0.016702723],"category_scores_gemma":[0.010009879,0.0006964845,0.0024174845,0.0017833267,0.0009402301,0.0035471658,0.0033909127,0.002854541,0.01072467],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009215368,0.00049206923,0.008495989,0.002083001,0.00027750392,0.0005357017,0.00043948402,0.016053203,0.004742091,0.0042433245,0.8583131,0.10340305],"study_design_scores_gemma":[0.0012318946,0.0007236805,0.038298804,0.0010690002,0.00025491405,0.001809606,0.0015800261,0.38445166,0.022822276,0.03994201,0.5074184,0.000397848],"about_ca_topic_score_codex":0.035373233,"about_ca_topic_score_gemma":0.06571553,"teacher_disagreement_score":0.035373233,"about_ca_system_score_codex":0.0018032264,"about_ca_system_score_gemma":0.0014800789,"threshold_uncertainty_score":0.07033467},"labels":[],"label_agreement":null},{"id":"W4416036439","doi":"10.18653/v1/2025.emnlp-main.567","title":"Back Attention: Understanding and Enhancing Multi-Hop Reasoning in Large Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Azrieli Foundation; European Commission; University of Manchester; Open Philanthropy Project","keywords":"Natural language; Language model; Feature (linguistics); Semantics (computer science); Language understanding; Automated reasoning","score_opus":0.03062317003061937,"score_gpt":0.3144617111784772,"score_spread":0.28383854114785784,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416036439","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10480004,0.000574092,0.8839436,0.0008420924,0.00007258082,0.00010480463,0.000276252,0.007142198,0.0022442993],"genre_scores_gemma":[0.83684015,0.0002086707,0.15939896,0.00039625014,0.00005634482,0.00008621962,0.00056718144,0.00025678688,0.0021893638],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991398,0.00031024584,0.000044897268,0.00025145544,0.00013931525,0.00011428032],"domain_scores_gemma":[0.9952663,0.00362824,0.0002560748,0.00045869034,0.0002486703,0.00014206745],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002587082,0.0015514031,0.00077977986,0.0011628551,0.0006691524,0.0018313428,0.0017903075,0.0012092999,0.0028361953],"category_scores_gemma":[0.010362135,0.0006541677,0.0012636195,0.0006366255,0.0008271758,0.005492167,0.0025301003,0.0025829328,0.0007981966],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006634669,0.00045824534,0.009077235,0.0002610784,0.00021111887,0.00050492724,0.0013995654,0.34620854,0.01700996,0.017448172,0.005706571,0.6010511],"study_design_scores_gemma":[0.000014552547,0.000036666734,0.00029982236,0.000009453877,0.000023116465,0.000021309457,0.000045402205,0.9810161,0.002978316,0.015178385,0.0003679643,0.000008961205],"about_ca_topic_score_codex":0.015293718,"about_ca_topic_score_gemma":0.023410931,"teacher_disagreement_score":0.015293718,"about_ca_system_score_codex":0.0015589471,"about_ca_system_score_gemma":0.0015389147,"threshold_uncertainty_score":0.030409396},"labels":[],"label_agreement":null},{"id":"W4416036623","doi":"10.18653/v1/2025.emnlp-main.607","title":"ChartGaze: Enhancing Chart Understanding in LVLMs with Eye-Tracking Guided Attention Refinement","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Chart; Natural language; Natural (archaeology); Empirical research; Natural language generation","score_opus":0.0389669361203135,"score_gpt":0.3372278773477799,"score_spread":0.29826094122746644,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416036623","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03220273,0.0027372225,0.8990591,0.00037450108,0.0004337876,0.00031014945,0.0028593133,0.056998625,0.0050245314],"genre_scores_gemma":[0.36498332,0.0014829531,0.6071139,0.0006518808,0.00023901409,0.000505491,0.008921812,0.0025791053,0.01352253],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99967635,0.00007067758,0.00001674841,0.00013205972,0.00006666955,0.000037568454],"domain_scores_gemma":[0.9994142,0.0003037555,0.00003736069,0.00007081759,0.00013530959,0.000038505128],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006282521,0.0015933258,0.0009753128,0.0008245061,0.0002921776,0.0012514294,0.0016621025,0.0010440244,0.010869309],"category_scores_gemma":[0.002507429,0.0003095808,0.0006969357,0.0005771038,0.00025742283,0.002104939,0.0014488582,0.0013776289,0.0038006324],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005705635,0.00014895723,0.0006014168,0.00045600915,0.00010322287,0.00018310804,0.00028284077,0.021216957,0.06176557,0.0019533797,0.03544095,0.877277],"study_design_scores_gemma":[0.00016043891,0.00031883243,0.0017812332,0.00008233335,0.000096381904,0.00013306564,0.00019846794,0.9303861,0.043503053,0.0062436215,0.017036812,0.00005947829],"about_ca_topic_score_codex":0.008576529,"about_ca_topic_score_gemma":0.0145148225,"teacher_disagreement_score":0.010869309,"about_ca_system_score_codex":0.00053685147,"about_ca_system_score_gemma":0.00065129955,"threshold_uncertainty_score":0.036361396},"labels":[],"label_agreement":null},{"id":"W4416037147","doi":"10.18653/v1/2025.emnlp-industry.187","title":"GEAR: A Scalable and Interpretable Evaluation Framework for RAG-Based Car Assistant Systems","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Scalability; Natural language; Natural (archaeology); Empirical research; Scale (ratio); Component (thermodynamics)","score_opus":0.02182292723303706,"score_gpt":0.34061300225866864,"score_spread":0.31879007502563156,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416037147","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010853338,0.00086189795,0.9026901,0.00036508666,0.00018615567,0.0007024299,0.003152382,0.07615245,0.005036078],"genre_scores_gemma":[0.29857463,0.0005280375,0.67467064,0.00038541615,0.00020917835,0.0012623024,0.01021092,0.004328386,0.009830418],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99589807,0.0011199792,0.0003627755,0.00069166534,0.0016364218,0.000291025],"domain_scores_gemma":[0.99690443,0.001157829,0.00018466648,0.0005842654,0.0010082636,0.00016059747],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043519028,0.0020598911,0.0017137579,0.0021337771,0.0009222421,0.00275029,0.0037934356,0.0014737794,0.012054092],"category_scores_gemma":[0.010239681,0.0006696758,0.0015388619,0.0008681692,0.0008476655,0.004928662,0.0039133513,0.0016560095,0.0042118323],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023984273,0.00071214075,0.0038083878,0.0012530229,0.0005344033,0.00065867935,0.00064900226,0.13918003,0.020937897,0.047080882,0.1174535,0.6653337],"study_design_scores_gemma":[0.0001818308,0.00021699336,0.0009566361,0.000094673705,0.00010613615,0.00014637334,0.00014961045,0.91772467,0.010720256,0.038634587,0.030988902,0.000079398196],"about_ca_topic_score_codex":0.012023551,"about_ca_topic_score_gemma":0.020078281,"teacher_disagreement_score":0.012054092,"about_ca_system_score_codex":0.0014934017,"about_ca_system_score_gemma":0.0018221117,"threshold_uncertainty_score":0.040324926},"labels":[],"label_agreement":null},{"id":"W4416037383","doi":"10.18653/v1/2025.emnlp-demos.68","title":"From Behavioral Performance to Internal Competence: Interpreting Vision-Language Models with VLM-Lens","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Natural (archaeology); Empirical research; Natural language; Action (physics)","score_opus":0.01081601759030118,"score_gpt":0.3170005470036598,"score_spread":0.3061845294133586,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416037383","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5252227,0.0017533813,0.42833805,0.004995226,0.0002923237,0.0002812042,0.0056837895,0.006142809,0.027290473],"genre_scores_gemma":[0.9749108,0.00015367994,0.022867953,0.0001456405,0.000017122104,0.00006665708,0.00070236356,0.00027063114,0.0008652898],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99914014,0.00044058062,0.000032327815,0.00020052984,0.000096525546,0.00008990789],"domain_scores_gemma":[0.99299276,0.0044087064,0.00069663423,0.000850739,0.0007358328,0.00031538884],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030773117,0.0007556032,0.0004786758,0.0009874541,0.00029022616,0.0034273604,0.0010878682,0.0008899147,0.005839454],"category_scores_gemma":[0.032797728,0.00053671846,0.0006878377,0.0006979041,0.00088400097,0.0037494937,0.0022615197,0.0016562972,0.0015795572],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002316091,0.00079640537,0.15780874,0.0009023143,0.0006967845,0.0008436683,0.007289207,0.09065472,0.033931967,0.050298948,0.03138271,0.6230784],"study_design_scores_gemma":[0.00014034024,0.00029513738,0.06026416,0.00023156386,0.00018860186,0.0002890813,0.0028241505,0.76209474,0.010358212,0.15877423,0.004370769,0.00016902326],"about_ca_topic_score_codex":0.009003407,"about_ca_topic_score_gemma":0.008384373,"teacher_disagreement_score":0.009003407,"about_ca_system_score_codex":0.0010017963,"about_ca_system_score_gemma":0.0006996081,"threshold_uncertainty_score":0.019534945},"labels":[],"label_agreement":null},{"id":"W4416047703","doi":"10.48550/arxiv.2505.21979","title":"Pearl: A Multimodal Culturally-Aware Arabic Instruction Dataset","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Pearl; Benchmark (surveying); Arabic; Workflow; Benchmarking; Mainstream; Semantics (computer science)","score_opus":0.03124253345628599,"score_gpt":0.31305448999951957,"score_spread":0.2818119565432336,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416047703","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16544896,0.0043562474,0.069473095,0.0038340753,0.0011423986,0.0018081691,0.6437037,0.047099955,0.063133456],"genre_scores_gemma":[0.1544408,0.00089919707,0.08348548,0.0010441827,0.0001061181,0.0021575852,0.74075186,0.0011331005,0.015981678],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992854,0.00022750991,0.00006390303,0.00020849546,0.00015319005,0.00006159224],"domain_scores_gemma":[0.99888617,0.0003738164,0.000056826455,0.00028448313,0.0003047591,0.00009398012],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007581659,0.0019654704,0.0005103844,0.0018365363,0.0010600686,0.0012508186,0.0020595011,0.0019401957,0.015749117],"category_scores_gemma":[0.0053495197,0.0002864567,0.0007867879,0.0016946775,0.00061556627,0.002021002,0.002476267,0.0020538815,0.011086465],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007035912,0.0007050834,0.009924238,0.0022548076,0.00014208109,0.001023877,0.0012204099,0.016117191,0.008454072,0.0069020498,0.66962206,0.28293046],"study_design_scores_gemma":[0.00033293618,0.0004939298,0.02478307,0.000913809,0.00014517245,0.001351979,0.0041421936,0.15047102,0.024693197,0.020405525,0.7719477,0.00031947397],"about_ca_topic_score_codex":0.018042611,"about_ca_topic_score_gemma":0.032044757,"teacher_disagreement_score":0.018042611,"about_ca_system_score_codex":0.0014834505,"about_ca_system_score_gemma":0.0014209719,"threshold_uncertainty_score":0.052686095},"labels":[],"label_agreement":null},{"id":"W4416052645","doi":"10.1109/iccv51701.2025.00693","title":"TerraMind: Large-Scale Generative Multimodality for Earth Observation","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Impact","funders":"Gauss Centre for Supercomputing; European Space Agency","keywords":"Multimodality; Inference; Geospatial analysis; Generative grammar; Security token; Semantics (computer science); Modalities; Multimodal learning; Earth observation","score_opus":0.038109397919419856,"score_gpt":0.33133000540431273,"score_spread":0.29322060748489287,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416052645","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029601801,0.00077850395,0.9460011,0.0009113932,0.0001682811,0.00013242765,0.0055583287,0.011679759,0.0051683895],"genre_scores_gemma":[0.57220405,0.0007626628,0.38608444,0.0013544551,0.00019725708,0.00047990258,0.021323308,0.0022373942,0.015356508],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99966943,0.00008276752,0.000011586234,0.00014420574,0.000053756095,0.00003834213],"domain_scores_gemma":[0.9994342,0.0002697404,0.000036587753,0.00015897816,0.000055431894,0.00004506585],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008852109,0.001195286,0.00068926235,0.00073524175,0.00046068867,0.0011020472,0.0025451472,0.0013406338,0.0069685094],"category_scores_gemma":[0.003071974,0.0007828007,0.0020922062,0.00070494524,0.00080144074,0.0019753377,0.0032715609,0.0028911175,0.0021812816],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003938534,0.00024230519,0.00804371,0.00035099118,0.0005520329,0.00035641226,0.00041970663,0.59094197,0.01008544,0.030173287,0.03464869,0.32379162],"study_design_scores_gemma":[0.000015171067,0.000037195838,0.00072088686,0.000025373633,0.000026240969,0.000076215365,0.000028099446,0.97480494,0.0017243945,0.017049542,0.005470653,0.000021369166],"about_ca_topic_score_codex":0.012810005,"about_ca_topic_score_gemma":0.028957833,"teacher_disagreement_score":0.012810005,"about_ca_system_score_codex":0.001046633,"about_ca_system_score_gemma":0.0008000695,"threshold_uncertainty_score":0.025470853},"labels":[],"label_agreement":null},{"id":"W4416098793","doi":"10.1109/iccv51701.2025.00626","title":"Placeit3d: Language-Guided Object Placement in Real 3D Scenes","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Task (project management); Suite; Benchmark (surveying); Object (grammar); Asset (computer security); Point (geometry); 3d model","score_opus":0.024564468975065434,"score_gpt":0.34878465117733887,"score_spread":0.32422018220227344,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416098793","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.058575343,0.0016386051,0.8166691,0.0010987994,0.00053130183,0.0005646797,0.012606715,0.093805045,0.01451031],"genre_scores_gemma":[0.2757994,0.00059389713,0.68346626,0.00093669567,0.00009338454,0.0005693257,0.024971172,0.004216833,0.009353023],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988166,0.00027115404,0.0000645475,0.00048067092,0.00023435605,0.00013278358],"domain_scores_gemma":[0.9986487,0.00046893756,0.00008422482,0.0005310042,0.0001530005,0.000114275805],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011131458,0.0030134174,0.0012394642,0.0012065058,0.0008569158,0.002855877,0.004928774,0.0037398909,0.011075482],"category_scores_gemma":[0.0054397034,0.0008983441,0.002339881,0.0013047091,0.0015839414,0.0042273942,0.004026878,0.0025044149,0.007330269],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011439999,0.0006017057,0.0061138645,0.0019019581,0.0002642585,0.0010393757,0.00096594053,0.29754266,0.030619703,0.02459534,0.14356424,0.49164695],"study_design_scores_gemma":[0.00013896858,0.0002569914,0.001046702,0.00015204196,0.000033162916,0.00045370957,0.0005035975,0.91187745,0.023603076,0.026074324,0.03578301,0.00007689153],"about_ca_topic_score_codex":0.0123550305,"about_ca_topic_score_gemma":0.024008427,"teacher_disagreement_score":0.0123550305,"about_ca_system_score_codex":0.0018972217,"about_ca_system_score_gemma":0.0017915426,"threshold_uncertainty_score":0.03705114},"labels":[],"label_agreement":null},{"id":"W4416118802","doi":"10.48550/arxiv.2504.05227","title":"A Reality Check of Vision-Language Pre-training in Radiology: Have We Progressed Using Text?","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Popularity; Feature (linguistics); Reality check; Check-in; Noise (video); Visualization","score_opus":0.06354629236181726,"score_gpt":0.38898361299617124,"score_spread":0.325437320634354,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416118802","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3023341,0.035253357,0.5547174,0.041472908,0.005315945,0.0007919144,0.005468792,0.013889058,0.040756516],"genre_scores_gemma":[0.72486305,0.0037340226,0.24146561,0.00633259,0.00127442,0.00039993852,0.009412276,0.0015885335,0.010929552],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99123317,0.0047579138,0.00036308105,0.0022577522,0.0009306592,0.0004573527],"domain_scores_gemma":[0.968386,0.020270316,0.0009620422,0.006107906,0.003307924,0.00096568774],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017531965,0.002758156,0.0015158311,0.0010966606,0.0018602385,0.0038812638,0.0032991294,0.0045557306,0.009380765],"category_scores_gemma":[0.07036319,0.0007769106,0.0012378207,0.0010079512,0.0029576821,0.013434359,0.0037951134,0.008049943,0.005830188],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0043320027,0.001858128,0.013642637,0.002725912,0.00052396616,0.00047312025,0.0011860701,0.08045626,0.015375461,0.02245566,0.06734724,0.78962356],"study_design_scores_gemma":[0.0006126585,0.0031679054,0.010560848,0.0016547601,0.00035336424,0.00093486375,0.0019906382,0.73379743,0.051243715,0.13300478,0.06239153,0.00028741738],"about_ca_topic_score_codex":0.0066863615,"about_ca_topic_score_gemma":0.009324841,"teacher_disagreement_score":0.017531965,"about_ca_system_score_codex":0.0017477119,"about_ca_system_score_gemma":0.00222101,"threshold_uncertainty_score":0.09271896},"labels":[],"label_agreement":null},{"id":"W4416183485","doi":"10.1109/mipr67560.2025.00054","title":"Mitigating Image Captioning Hallucinations in Vision-Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Closed captioning; Retraining; Normalization (sociology); Inference; Reliability (semiconductor); Adaptation (eye); Test data; Reduction (mathematics)","score_opus":0.009100288445900715,"score_gpt":0.32291833367351613,"score_spread":0.3138180452276154,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416183485","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06794984,0.0010004237,0.92480904,0.00046897816,0.00018290429,0.0001092658,0.0001720716,0.0038211832,0.00148625],"genre_scores_gemma":[0.7079877,0.0007985175,0.2831698,0.0009641578,0.00022327216,0.00017698888,0.0010886342,0.000573929,0.005016962],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982551,0.00057026226,0.00009333028,0.00043296118,0.00046824515,0.00018016735],"domain_scores_gemma":[0.994769,0.0023311137,0.00044987162,0.0011426642,0.0010734132,0.00023394564],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030631968,0.0016499739,0.0012854652,0.0006704799,0.00053231505,0.0014604026,0.0020490326,0.0015506477,0.0013729817],"category_scores_gemma":[0.014474377,0.00063738227,0.0010726067,0.0006424424,0.0010579333,0.0027864142,0.0028028924,0.003011009,0.00085729396],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00091773743,0.0003289923,0.0028768047,0.0003333241,0.00028527353,0.00047595153,0.0004422762,0.33814713,0.047017135,0.0039127828,0.0071951505,0.59806746],"study_design_scores_gemma":[0.000017104423,0.000114835806,0.00047355995,0.000011197229,0.000033728516,0.00013796234,0.00003892327,0.9788167,0.01712686,0.002294713,0.0009136621,0.000020662925],"about_ca_topic_score_codex":0.0051981714,"about_ca_topic_score_gemma":0.006057022,"teacher_disagreement_score":0.0051981714,"about_ca_system_score_codex":0.0008946741,"about_ca_system_score_gemma":0.0011289454,"threshold_uncertainty_score":0.016199946},"labels":[],"label_agreement":null},{"id":"W4416237890","doi":"10.1007/978-3-032-08452-1_4","title":"Vision Language Models for Dynamic Human Activity Recognition in Healthcare Settings","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Flexibility (engineering); Benchmark (surveying); Key (lock); Activity recognition; Health care; Generative grammar; Deep learning; Code (set theory)","score_opus":0.016681572316009224,"score_gpt":0.32324759820422455,"score_spread":0.30656602588821535,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416237890","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029501904,0.0038353996,0.955116,0.0010781504,0.00032868338,0.00007510873,0.0014392888,0.0039173453,0.0047080596],"genre_scores_gemma":[0.7802224,0.003441401,0.18565342,0.00085363776,0.0003105578,0.00032205187,0.0036561969,0.00047894835,0.025061373],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999729,0.000062855,0.000015008714,0.00008702709,0.00005193137,0.000054119842],"domain_scores_gemma":[0.9995253,0.00028065598,0.000034572095,0.000037100384,0.00009913382,0.000023210672],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006443089,0.00072270515,0.0008947264,0.0006091215,0.00021992171,0.0010653965,0.0013301115,0.0009881624,0.004088821],"category_scores_gemma":[0.0019320365,0.00043653016,0.0010731596,0.0007186462,0.0002572435,0.0011300026,0.0005099353,0.0016584339,0.0028076463],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004156053,0.00034739333,0.0019558857,0.00022183654,0.00020228866,0.00023699398,0.00014224222,0.335745,0.012549586,0.019463738,0.020308424,0.608411],"study_design_scores_gemma":[0.0000062033514,0.000025324944,0.0003280479,0.000011666656,0.000018443345,0.000045970428,0.0000125549905,0.99154305,0.0009894458,0.005986682,0.0010226369,0.000009941509],"about_ca_topic_score_codex":0.016014414,"about_ca_topic_score_gemma":0.017073005,"teacher_disagreement_score":0.016014414,"about_ca_system_score_codex":0.00088767806,"about_ca_system_score_gemma":0.00075438013,"threshold_uncertainty_score":0.03184241},"labels":[],"label_agreement":null},{"id":"W4416252247","doi":"10.1109/ijcnn64981.2025.11228324","title":"GNN-ViTCap: GNN-Enhanced Multiple Instance Learning with Vision Transformers for Whole Slide Image Classification and Captioning","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Pattern recognition (psychology); Cluster analysis; Closed captioning; Feature extraction; Contextual image classification; Graph; Feature vector; Pixel; Artificial neural network","score_opus":0.00910167173983274,"score_gpt":0.28663465930706206,"score_spread":0.27753298756722933,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416252247","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025336023,0.0015801042,0.9173266,0.00051519566,0.0006301576,0.00044419372,0.0021676267,0.046078533,0.0059215412],"genre_scores_gemma":[0.28783602,0.0007702652,0.6751651,0.0018625368,0.00025831538,0.00080271094,0.014746805,0.0016654182,0.016892789],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992681,0.00013099876,0.00003410271,0.00030510835,0.00015832101,0.00010343582],"domain_scores_gemma":[0.9992493,0.00023594112,0.000057066067,0.00016722069,0.00022723085,0.00006318615],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012513429,0.002483597,0.0011993565,0.0012851463,0.0006082489,0.0013460341,0.0055697,0.002712069,0.007213113],"category_scores_gemma":[0.0030661193,0.00076600554,0.0019115966,0.0014535445,0.00062503625,0.0020870117,0.0019952527,0.0033249743,0.003536607],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000448286,0.00040294733,0.0011509683,0.00033659293,0.0002802898,0.0003138573,0.00007651539,0.21010794,0.014563036,0.003703714,0.0532354,0.7153805],"study_design_scores_gemma":[0.000022728333,0.000053856715,0.00017038919,0.000011361484,0.000022653589,0.00005848606,0.000012294879,0.9895627,0.005209367,0.0022219582,0.0026396727,0.00001454788],"about_ca_topic_score_codex":0.014177376,"about_ca_topic_score_gemma":0.0154739665,"teacher_disagreement_score":0.014177376,"about_ca_system_score_codex":0.0017945648,"about_ca_system_score_gemma":0.0013589239,"threshold_uncertainty_score":0.028189719},"labels":[],"label_agreement":null},{"id":"W4416365506","doi":"10.21203/rs.3.rs-7900022/v1","title":"GRAVITI: Grounded Retrieval Generation Framework for VideoLLM Hallucination Mitigation","year":2025,"lang":"","type":"preprint","venue":"Research Square","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Closed captioning; Security token; Process (computing); Decoding methods; Task (project management); Task analysis; Ground; Detector","score_opus":0.11216086555727185,"score_gpt":0.46550703184026554,"score_spread":0.3533461662829937,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416365506","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001232438,0.000096256226,0.9889249,0.00007935827,0.00004538517,0.00007676104,0.00038540107,0.008318049,0.00084146386],"genre_scores_gemma":[0.094449826,0.00023139776,0.89228654,0.00027410642,0.00011889659,0.00031715888,0.0027845514,0.0026991125,0.006838488],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994628,0.00012020242,0.0000236719,0.00012583814,0.00020470086,0.000062726474],"domain_scores_gemma":[0.99950254,0.00013049826,0.000031155196,0.00013839618,0.00016048053,0.000036995385],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010372334,0.0015772328,0.0009418269,0.0012919214,0.0006916671,0.0018630241,0.0027838966,0.0019095442,0.014176243],"category_scores_gemma":[0.0030806754,0.0005698123,0.0012515241,0.00094720424,0.0006964128,0.001738103,0.0031125897,0.0017759113,0.006086348],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00077313575,0.00017634773,0.00060431595,0.0005874402,0.0002071584,0.00048904517,0.00043206208,0.082872264,0.06334772,0.036275662,0.04695799,0.76727694],"study_design_scores_gemma":[0.00007999977,0.00013827573,0.00024423894,0.000042877142,0.000050449456,0.00023190644,0.00010077615,0.91083544,0.03389352,0.036767617,0.017564995,0.00004981702],"about_ca_topic_score_codex":0.0037649986,"about_ca_topic_score_gemma":0.006137196,"teacher_disagreement_score":0.014176243,"about_ca_system_score_codex":0.0007119081,"about_ca_system_score_gemma":0.0009188924,"threshold_uncertainty_score":0.047424257},"labels":[],"label_agreement":null},{"id":"W4416366377","doi":"10.1109/tvcg.2025.3634791","title":"Probing the Visualization Literacy of Vision Language Models: The Good, the Bad, and the Ugly","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Visualization and Computer Graphics","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Visualization; Chart; Correctness; Comprehension; Data visualization; Program comprehension; Code (set theory); Visual reasoning; Key (lock)","score_opus":0.009481657295951497,"score_gpt":0.30011784424893456,"score_spread":0.29063618695298304,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416366377","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.41826713,0.002226149,0.42774668,0.0056547564,0.00049909804,0.00069200614,0.009168637,0.10281911,0.0329264],"genre_scores_gemma":[0.84378403,0.00045572608,0.14124177,0.0007598286,0.000059314974,0.00037864107,0.006109722,0.003275753,0.003935247],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9978236,0.00096413575,0.0001167934,0.0005854096,0.0003703611,0.00013970956],"domain_scores_gemma":[0.98455215,0.011353586,0.0005815234,0.0020037526,0.0011556848,0.00035320572],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042103566,0.001460048,0.0006152839,0.0010322507,0.00041172482,0.0046072775,0.002431084,0.0018904352,0.0071542487],"category_scores_gemma":[0.04026228,0.00056446536,0.0014642056,0.0006469597,0.0012492744,0.0066830055,0.0024962325,0.0027417345,0.002016767],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0029127628,0.00080693123,0.038902394,0.003201066,0.0006548996,0.0010528081,0.010165364,0.16702403,0.048265927,0.052192666,0.093487605,0.58133364],"study_design_scores_gemma":[0.00024792136,0.0004803774,0.008297482,0.00030676278,0.00020636305,0.0003518619,0.0017193471,0.84228265,0.02745724,0.0783746,0.04012674,0.00014862938],"about_ca_topic_score_codex":0.008358584,"about_ca_topic_score_gemma":0.010425385,"teacher_disagreement_score":0.008358584,"about_ca_system_score_codex":0.0019395762,"about_ca_system_score_gemma":0.0014550816,"threshold_uncertainty_score":0.023933351},"labels":[],"label_agreement":null},{"id":"W4416401253","doi":"10.1109/ismar-adjunct68609.2025.00263","title":"Project LOCOMO AR: Augmented Reality with Carbon Metrics for Sustainable AI Use","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Algoma University","funders":"National Research Foundation","keywords":"Augmented reality; Carbon footprint; Visualization; Data visualization; Carbon fibers; Field (mathematics)","score_opus":0.028191111005435008,"score_gpt":0.33895225613586555,"score_spread":0.3107611451304305,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416401253","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04232251,0.0013142632,0.8658605,0.0017622149,0.0004970916,0.00051380065,0.010081891,0.047642585,0.030005155],"genre_scores_gemma":[0.37920007,0.0011441456,0.5979724,0.0005982834,0.000139546,0.00062506786,0.009401021,0.0028346733,0.008084815],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9991053,0.00027212367,0.000032451262,0.00017665021,0.00036387405,0.00004971612],"domain_scores_gemma":[0.99926645,0.00030735336,0.000048063248,0.00017079044,0.00015284639,0.000054608314],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010177667,0.001447202,0.0005068905,0.0008633956,0.0003592255,0.0020967058,0.0013596733,0.0010011285,0.009380163],"category_scores_gemma":[0.0028447001,0.00032263208,0.00081391126,0.0007266991,0.0004907243,0.0018094564,0.002447995,0.0012253106,0.0016371966],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011841233,0.0006152353,0.0045542787,0.001240355,0.0003781112,0.000652454,0.0012570043,0.08746818,0.062038407,0.036437064,0.15023862,0.6539361],"study_design_scores_gemma":[0.00022407796,0.00057909574,0.0048233555,0.0002470336,0.00014605944,0.0006732848,0.00072181236,0.69276124,0.045413204,0.040045626,0.21410793,0.0002572803],"about_ca_topic_score_codex":0.004996395,"about_ca_topic_score_gemma":0.007615182,"teacher_disagreement_score":0.009380163,"about_ca_system_score_codex":0.00059909315,"about_ca_system_score_gemma":0.00069714186,"threshold_uncertainty_score":0.03137976},"labels":[],"label_agreement":null},{"id":"W4416402720","doi":"10.1109/ismar-adjunct68609.2025.00223","title":"Region-Guided Interactive Docent for Paintings in Mixed Reality","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Interface","keywords":"Painting; Mixed reality; Cursor (databases); Visitor pattern; Augmented reality; Virtual reality","score_opus":0.03880225086164991,"score_gpt":0.3627388482090029,"score_spread":0.323936597347353,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416402720","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.056846056,0.0007393348,0.89691037,0.0003615148,0.00009370878,0.0001951186,0.00064172345,0.031598832,0.012613361],"genre_scores_gemma":[0.46504954,0.00043202448,0.51992446,0.000423716,0.000056289384,0.00023079822,0.0009907016,0.0012450553,0.011647461],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996314,0.0001071122,0.000011853667,0.000094715506,0.000098629345,0.000056257097],"domain_scores_gemma":[0.99959797,0.00019893525,0.000021402386,0.0000941046,0.000038289254,0.000049322545],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00044677444,0.0011291702,0.00052114454,0.000550673,0.0005252761,0.0013358326,0.0014082085,0.0011841023,0.0120947035],"category_scores_gemma":[0.0015063502,0.0004192219,0.0012092411,0.00023200695,0.0005578806,0.0012742967,0.0022920086,0.00086516526,0.00198682],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023930927,0.00068121584,0.0033709672,0.0015888115,0.0004112726,0.003796963,0.011425652,0.062166724,0.28480607,0.029884176,0.035876367,0.56359863],"study_design_scores_gemma":[0.00036176853,0.0007163717,0.0050529744,0.00020166862,0.0003190283,0.0036140024,0.0022242558,0.6569188,0.13199222,0.018154314,0.18004888,0.00039562446],"about_ca_topic_score_codex":0.0030530018,"about_ca_topic_score_gemma":0.00824845,"teacher_disagreement_score":0.0120947035,"about_ca_system_score_codex":0.0004157648,"about_ca_system_score_gemma":0.00036962275,"threshold_uncertainty_score":0.040460825},"labels":[],"label_agreement":null},{"id":"W4416417137","doi":"10.1007/s11432-025-4676-4","title":"Large multimodal models evaluation: a survey","year":2025,"lang":"en","type":"article","venue":"Science China Information Sciences","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Competence (human resources); Data collection; Evaluation methods; Term (time); Multimodal interaction","score_opus":0.03383829650789999,"score_gpt":0.36397139492180575,"score_spread":0.33013309841390576,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416417137","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.072433904,0.50408435,0.37004352,0.005699029,0.001331082,0.0007566923,0.0054488704,0.007571402,0.032631207],"genre_scores_gemma":[0.57458204,0.13896115,0.23631023,0.002880557,0.001954691,0.0010788098,0.023348007,0.0033084874,0.017575929],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9920855,0.0037367374,0.0005829742,0.0010569762,0.0023127568,0.00022495748],"domain_scores_gemma":[0.98313624,0.012350985,0.00039384255,0.0016445857,0.0021763963,0.00029805556],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.010946519,0.0031990784,0.004004644,0.0047084256,0.00087218935,0.0027158028,0.0046289098,0.0019154806,0.009576092],"category_scores_gemma":[0.034789536,0.0007605539,0.0018820472,0.0038760665,0.00079963694,0.0051156217,0.002763778,0.0014381395,0.0023174633],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005458355,0.0004119263,0.0045805024,0.0021436366,0.0005432622,0.00007450096,0.0000771598,0.024951104,0.0010261683,0.0032273335,0.034915227,0.9275033],"study_design_scores_gemma":[0.00034592475,0.0019040266,0.015992807,0.003087522,0.0020356833,0.0010970435,0.0008756125,0.7972913,0.010434882,0.04172902,0.12501821,0.00018795644],"about_ca_topic_score_codex":0.007573685,"about_ca_topic_score_gemma":0.008610157,"teacher_disagreement_score":0.9890535,"about_ca_system_score_codex":0.0017610624,"about_ca_system_score_gemma":0.0020852368,"threshold_uncertainty_score":0.05789137},"labels":[],"label_agreement":null},{"id":"W4416548821","doi":"10.1109/iccv51701.2025.01902","title":"Representation Shift: Unifying Token Compression with Flashattention","year":2025,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Security token; Data compression; Compression (physics); Computation; Transformer; Representation (politics); Overhead (engineering); Lossless compression; Decoding methods","score_opus":0.013642536399723782,"score_gpt":0.3111054682249951,"score_spread":0.2974629318252713,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416548821","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04242022,0.00048806093,0.9424393,0.00034406348,0.00016490063,0.00013481548,0.0003150161,0.011182348,0.0025113237],"genre_scores_gemma":[0.5869925,0.00044360536,0.40129396,0.00045904724,0.00010740145,0.00021812003,0.0012924738,0.0014029255,0.0077899764],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996063,0.00006728481,0.000031847238,0.000113151335,0.00012177201,0.00005971147],"domain_scores_gemma":[0.9989483,0.00034613564,0.00008393187,0.00039903677,0.00015443133,0.00006821895],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006807843,0.0009107605,0.00061963283,0.0008507878,0.0003359924,0.0010608834,0.0018736325,0.0006412954,0.004561528],"category_scores_gemma":[0.003802747,0.00031213724,0.0005967885,0.0007850085,0.0008831944,0.0030808158,0.0025363234,0.0013439173,0.0017302213],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00057181384,0.00018246663,0.0017911709,0.00017927661,0.00006579607,0.00022815789,0.00029026213,0.06823305,0.03994185,0.019408247,0.010764762,0.85834324],"study_design_scores_gemma":[0.000057293128,0.00022289446,0.00067500374,0.000032899956,0.00005863877,0.00026020064,0.00009364641,0.8857764,0.07586276,0.027654376,0.009269214,0.000036691352],"about_ca_topic_score_codex":0.0035190398,"about_ca_topic_score_gemma":0.004803785,"teacher_disagreement_score":0.004561528,"about_ca_system_score_codex":0.00087409036,"about_ca_system_score_gemma":0.0011721262,"threshold_uncertainty_score":0.015259802},"labels":[],"label_agreement":null},{"id":"W4416551090","doi":"10.48550/arxiv.2504.12083","title":"Self-alignment of Large Video Language Models with Refined Regularized Preference Optimization","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Mitacs","keywords":"Spurious relationship; Regularization (linguistics); Preference; Set (abstract data type); Simple (philosophy); Optimization problem","score_opus":0.026327536959017347,"score_gpt":0.2761713555977348,"score_spread":0.24984381863871746,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416551090","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027449721,0.00014792861,0.9695237,0.00016353822,0.000037512305,0.000060979597,0.00013360323,0.0017248298,0.0007582357],"genre_scores_gemma":[0.64096594,0.00017124056,0.3503214,0.00063707895,0.00009159084,0.0004433997,0.001288892,0.0009555205,0.0051248763],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99886096,0.00047699947,0.00006305464,0.00032277184,0.00016234942,0.00011383114],"domain_scores_gemma":[0.9976598,0.0013024238,0.00019640147,0.0002794656,0.0004207159,0.00014130275],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001992484,0.0012929424,0.0010749389,0.00053973653,0.00041941414,0.00089641986,0.001943999,0.0013052599,0.002468449],"category_scores_gemma":[0.007887932,0.0006871799,0.0010538021,0.0006371393,0.0007037793,0.0019373788,0.001726339,0.0021569375,0.00115606],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033827254,0.00014685855,0.001500294,0.00014372647,0.000089278314,0.00015247575,0.00020297292,0.8162304,0.008620819,0.009317501,0.00403192,0.15922548],"study_design_scores_gemma":[0.000007641155,0.000024205207,0.000045866396,0.0000028121904,0.0000027806816,0.000009516966,0.0000082028855,0.996563,0.00079804554,0.002375715,0.00015762584,0.0000045246798],"about_ca_topic_score_codex":0.0056514023,"about_ca_topic_score_gemma":0.007261258,"teacher_disagreement_score":0.0056514023,"about_ca_system_score_codex":0.0009352334,"about_ca_system_score_gemma":0.0013482897,"threshold_uncertainty_score":0.011237025},"labels":[],"label_agreement":null},{"id":"W4416726017","doi":"10.1109/igarss55030.2025.11243055","title":"CSA Mamba: A Channel-Spatial Attention Mamba Network for Image Captioning","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Innovation Fund; Nanjing University of Aeronautics and Astronautics; National Natural Science Foundation of China; State Key Laboratory of Integrated Services Networks; Natural Science Foundation of Jiangsu Province; Ministry of Natural Resources","keywords":"Closed captioning; Feature (linguistics); Image (mathematics); Task (project management); Natural language; Natural (archaeology); Perception","score_opus":0.011025088727546225,"score_gpt":0.2918848238776111,"score_spread":0.2808597351500649,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416726017","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02408641,0.0013826884,0.9577271,0.00045829653,0.00027278677,0.0002748324,0.0005537379,0.0090003805,0.0062438417],"genre_scores_gemma":[0.3978311,0.0009856849,0.58197284,0.00075835,0.00024302411,0.0005445816,0.0025486497,0.00067587255,0.014439943],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9996979,0.00008233334,0.0000114801605,0.00008791477,0.000069309404,0.00005102578],"domain_scores_gemma":[0.99953103,0.00014603767,0.000036357236,0.00009718015,0.00014777828,0.00004169025],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006035549,0.0011832094,0.0007928605,0.0010477643,0.00070125854,0.0007155075,0.0016961923,0.001179397,0.0069524976],"category_scores_gemma":[0.0017913285,0.00037698238,0.0007379791,0.00089855905,0.00056051626,0.0016258438,0.0015643262,0.0014334325,0.0017376762],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006054915,0.0002493999,0.0011030268,0.0003551446,0.00012226673,0.00020309856,0.0002241122,0.069058456,0.066469215,0.010726192,0.024711238,0.8261725],"study_design_scores_gemma":[0.000033057793,0.000130092,0.0009045361,0.000025618832,0.000050421004,0.00012957184,0.00008798352,0.93676394,0.033370744,0.010358623,0.01810614,0.000039230865],"about_ca_topic_score_codex":0.0057466133,"about_ca_topic_score_gemma":0.008546682,"teacher_disagreement_score":0.0069524976,"about_ca_system_score_codex":0.0008942488,"about_ca_system_score_gemma":0.0009703855,"threshold_uncertainty_score":0.023258388},"labels":[],"label_agreement":null},{"id":"W4416748856","doi":"10.1109/iros60139.2025.11246761","title":"Refined Policy Distillation: From VLA Generalists to RL Experts","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Ministry of Education","keywords":"Reinforcement learning; Complement (music); Policy learning; Generalization; Perspective (graphical); Key (lock); Task (project management)","score_opus":0.012370089613082181,"score_gpt":0.33520705237334264,"score_spread":0.32283696276026047,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416748856","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032680724,0.00067452126,0.9514704,0.00072881137,0.00018659749,0.00023349325,0.00028729255,0.008639494,0.0050986847],"genre_scores_gemma":[0.5795216,0.00035787738,0.40833563,0.0012594028,0.00011528744,0.0007500713,0.001065483,0.0016688802,0.006925785],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985689,0.0005343268,0.000073862204,0.0004166111,0.00022828295,0.00017803312],"domain_scores_gemma":[0.9958924,0.0024367424,0.00022674391,0.00090948486,0.0002966843,0.00023788745],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028160869,0.0018607599,0.0015306977,0.00073449133,0.0006344117,0.0015163104,0.0032173328,0.0018642757,0.005874827],"category_scores_gemma":[0.012559855,0.0011169295,0.0012340313,0.0004563435,0.0017093722,0.0023545397,0.0037586922,0.005168702,0.0018725655],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004079239,0.00023855154,0.0019949793,0.00024964343,0.00010512682,0.00015827783,0.00022433278,0.78335065,0.0042645507,0.020069757,0.0074360864,0.18150014],"study_design_scores_gemma":[0.000041971794,0.00004163349,0.000057038702,0.00001717534,0.000007829844,0.00001482602,0.000010422338,0.99046564,0.00093381334,0.0071994527,0.0012010692,0.000009198577],"about_ca_topic_score_codex":0.0065470203,"about_ca_topic_score_gemma":0.009019329,"teacher_disagreement_score":0.0065470203,"about_ca_system_score_codex":0.0014824365,"about_ca_system_score_gemma":0.0027715508,"threshold_uncertainty_score":0.019653201},"labels":[],"label_agreement":null},{"id":"W4416749111","doi":"10.1109/iros60139.2025.11247593","title":"OpenNav: Open-World Navigation with Multimodal Large Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Rehabilitation Institute; University of Toronto","funders":"","keywords":"Robustness (evolution); Leverage (statistics); Robot; Natural language; Semantic mapping; Mobile robot navigation; Bridging (networking); Perception","score_opus":0.014743656412547906,"score_gpt":0.3263966893968688,"score_spread":0.3116530329843209,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416749111","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009852344,0.00035806632,0.91136897,0.00025628312,0.00012221589,0.00014744524,0.0017976286,0.07314278,0.0029542446],"genre_scores_gemma":[0.23956235,0.00043528213,0.73684746,0.0007329621,0.000060402228,0.0007118096,0.010500649,0.0057163667,0.0054327645],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99952793,0.00010363404,0.000028200988,0.0001823047,0.000117367825,0.000040493633],"domain_scores_gemma":[0.9991887,0.0003298623,0.000059699447,0.0002049623,0.00015504718,0.00006165128],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00073818024,0.0016074858,0.0005798369,0.00048381585,0.00042548918,0.001568554,0.002829373,0.0013189134,0.007639086],"category_scores_gemma":[0.00384908,0.00073554064,0.0014318158,0.00030993784,0.000754949,0.0030702103,0.0030347968,0.0020271759,0.0037357162],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00084286276,0.00047048737,0.003446339,0.0010875214,0.00037467177,0.0009493503,0.0012879465,0.3096454,0.043200947,0.0359125,0.06995529,0.53282666],"study_design_scores_gemma":[0.000069732094,0.00009832929,0.00029378614,0.0000577691,0.000033092983,0.00011414831,0.00011594113,0.9502379,0.008589281,0.022837231,0.017489418,0.00006329078],"about_ca_topic_score_codex":0.009760205,"about_ca_topic_score_gemma":0.01827032,"teacher_disagreement_score":0.009760205,"about_ca_system_score_codex":0.00066806405,"about_ca_system_score_gemma":0.0013448577,"threshold_uncertainty_score":0.025555253},"labels":[],"label_agreement":null},{"id":"W4416749249","doi":"10.1109/iros60139.2025.11247596","title":"MORE: Mobile Manipulation Rearrangement Through Grounded Language Reasoning","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Toyota Motor Europe","keywords":"Object (grammar); Scheme (mathematics); Code (set theory); Set (abstract data type); Mobile device; Mobile robot; Foundation (evidence)","score_opus":0.01470210709137515,"score_gpt":0.334111618908203,"score_spread":0.31940951181682786,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416749249","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02614366,0.0007369432,0.9100361,0.0007447002,0.00016108733,0.00030063867,0.0022435186,0.047332764,0.012300667],"genre_scores_gemma":[0.37809473,0.00048239116,0.6040766,0.00058242166,0.000056457106,0.0003418815,0.006305782,0.0028803064,0.007179417],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9992169,0.00014884949,0.000040195857,0.0002177467,0.00026174096,0.00011467182],"domain_scores_gemma":[0.999171,0.00038231496,0.00005849405,0.0002476762,0.000087443696,0.000053166263],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00062867044,0.0017660544,0.00086565147,0.00064041314,0.0008264388,0.0019266126,0.0036455921,0.0017226811,0.009873584],"category_scores_gemma":[0.0027836545,0.0007373913,0.0020906774,0.00048623132,0.0014106977,0.003030398,0.0035535085,0.002184334,0.0020121904],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048127343,0.00033556734,0.0010991461,0.00082842103,0.00016761885,0.0007720644,0.00043153248,0.64904606,0.021624045,0.04427625,0.033579115,0.2473589],"study_design_scores_gemma":[0.00007743221,0.00007157076,0.00013701078,0.000033370416,0.000023012384,0.00009836738,0.0000945007,0.9618481,0.004990563,0.024701934,0.007899212,0.00002495499],"about_ca_topic_score_codex":0.013045958,"about_ca_topic_score_gemma":0.027382018,"teacher_disagreement_score":0.013045958,"about_ca_system_score_codex":0.0013405636,"about_ca_system_score_gemma":0.0019901833,"threshold_uncertainty_score":0.03303045},"labels":[],"label_agreement":null},{"id":"W4416749785","doi":"10.1109/iros60139.2025.11246581","title":"TagGuideBot: Enhancing Robot Intelligence with Object Tags and VLMs","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Artificial Intelligence in Medicine (Canada)","funders":"National Natural Science Foundation of China","keywords":"Robot; Object (grammar); Naturalness; Point (geometry); Motion (physics); Semantics (computer science); Semantic mapping; Visualization; SMT placement equipment","score_opus":0.009074963396594306,"score_gpt":0.2836824222708823,"score_spread":0.274607458874288,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416749785","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.046065502,0.00041402646,0.9296658,0.00013665408,0.00008006228,0.00012186822,0.00010775448,0.01978843,0.0036198595],"genre_scores_gemma":[0.4499762,0.00021309617,0.5415838,0.00035274826,0.000025074823,0.00018421195,0.00061428256,0.00079334236,0.0062572565],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99954754,0.000101896425,0.000015321013,0.00011753635,0.00016098641,0.000056644876],"domain_scores_gemma":[0.9994999,0.00020178476,0.00005439666,0.00011794294,0.00007542519,0.00005046491],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00066408707,0.00097148446,0.0005652991,0.00045886633,0.0002769613,0.00088150875,0.0018727409,0.00090812746,0.0020762952],"category_scores_gemma":[0.0016444918,0.00034072937,0.00045172253,0.00031885566,0.0008867499,0.0020346853,0.0016479864,0.00088552636,0.0011711881],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006198063,0.00065945357,0.00471428,0.00051990454,0.00011119739,0.000606441,0.0011608812,0.092689864,0.16094096,0.018028827,0.014114698,0.7058337],"study_design_scores_gemma":[0.00005232004,0.00048374583,0.0011135277,0.00003989518,0.00004427527,0.00034643803,0.00021000169,0.9018308,0.0650622,0.008378423,0.022364555,0.00007387545],"about_ca_topic_score_codex":0.0039226557,"about_ca_topic_score_gemma":0.006294753,"teacher_disagreement_score":0.0039226557,"about_ca_system_score_codex":0.00045842276,"about_ca_system_score_gemma":0.0009121547,"threshold_uncertainty_score":0.0077996254},"labels":[],"label_agreement":null},{"id":"W4416750376","doi":"10.1109/iros60139.2025.11247280","title":"Interpreting Behaviors and Geometric Constraints as Knowledge Graphs for Robot Manipulation Control","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Robot; Visual servoing; Scripting language; Flexibility (engineering); Control (management); Robot control; Semantics (computer science); Personal robot","score_opus":0.014274494479371737,"score_gpt":0.32774743408913143,"score_spread":0.3134729396097597,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416750376","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030563377,0.00017053573,0.9631339,0.000488965,0.000042396252,0.00011663113,0.0003056553,0.0011854821,0.003993085],"genre_scores_gemma":[0.479222,0.00029972458,0.51736534,0.00021369921,0.000027059143,0.00019666215,0.0006497069,0.0002893808,0.0017363918],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992755,0.00029088045,0.00004291404,0.00016430797,0.00017579792,0.000050660805],"domain_scores_gemma":[0.99740183,0.001629436,0.00025472604,0.0004301436,0.00020667548,0.000077275916],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008774227,0.0009627916,0.00036791313,0.00090999925,0.0004912503,0.0014805413,0.0012382069,0.0010652076,0.0026905253],"category_scores_gemma":[0.0051905774,0.00046795636,0.0008724399,0.0005169969,0.0018839631,0.003969631,0.0013951515,0.001443261,0.0003514743],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025741998,0.00020665172,0.002075007,0.00054952514,0.00012070268,0.0009880402,0.0021112573,0.46174794,0.026807163,0.33442497,0.0032280248,0.16748331],"study_design_scores_gemma":[0.00004645385,0.00007536789,0.0009062455,0.000073538846,0.00007068722,0.00014606699,0.0004776861,0.63674146,0.008818823,0.3408427,0.011745477,0.000055477743],"about_ca_topic_score_codex":0.0065558283,"about_ca_topic_score_gemma":0.010035731,"teacher_disagreement_score":0.0065558283,"about_ca_system_score_codex":0.0011150282,"about_ca_system_score_gemma":0.0009938811,"threshold_uncertainty_score":0.013035357},"labels":[],"label_agreement":null},{"id":"W4416817077","doi":"10.1016/j.compbiomed.2025.111347","title":"Med-VCD: Mitigating hallucination for medical large vision language models through visual contrastive decoding","year":2025,"lang":"en","type":"article","venue":"Computers in Biology and Medicine","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Decoding methods; Context (archaeology); Redundancy (engineering); Hallucinating; Modalities; Language model","score_opus":0.011914720684248828,"score_gpt":0.40482713905938916,"score_spread":0.39291241837514035,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416817077","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015141681,0.0007394457,0.9743066,0.0007038564,0.00015107243,0.00010311159,0.0005409524,0.006901031,0.0014122552],"genre_scores_gemma":[0.42288736,0.0005880341,0.5627951,0.0016841219,0.00029397092,0.00035375118,0.0034596296,0.0014913678,0.006446707],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9987251,0.00054839166,0.000084006235,0.0002768717,0.00028217075,0.00008341548],"domain_scores_gemma":[0.9946942,0.0039770994,0.0002037543,0.0005208371,0.00044869742,0.00015545178],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002507565,0.0014404545,0.000899587,0.0008881658,0.00043693103,0.0014740644,0.0021438815,0.0015224091,0.0039675334],"category_scores_gemma":[0.014461914,0.00052437163,0.0009972863,0.0005802684,0.0011808302,0.0016144856,0.0030444723,0.0026720949,0.0018353803],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008447999,0.00021355497,0.0024836587,0.0006096654,0.00021033817,0.0004929698,0.00040036388,0.29446596,0.023075312,0.01491653,0.02209013,0.6401966],"study_design_scores_gemma":[0.000054966476,0.00008480216,0.00013318092,0.000024865683,0.000021204687,0.00012462045,0.00003852834,0.9753804,0.010381696,0.011071421,0.0026651698,0.000019204543],"about_ca_topic_score_codex":0.0041827466,"about_ca_topic_score_gemma":0.008472282,"teacher_disagreement_score":0.0041827466,"about_ca_system_score_codex":0.0008953472,"about_ca_system_score_gemma":0.001638573,"threshold_uncertainty_score":0.013272703},"labels":[],"label_agreement":null},{"id":"W4417002434","doi":"10.1109/tits.2025.3635279","title":"Peer Learning Approach to Unbiased Scene Graph Generation for Traffic Scene Understanding","year":2025,"lang":"","type":"article","venue":"IEEE Transactions on Intelligent Transportation Systems","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Shenzhen Research Foundation; Chuzhou Science and Technology Program","keywords":"Graph; Boosting (machine learning); Scene graph; Peer-to-peer; Voting; Graph theory","score_opus":0.08458328145002182,"score_gpt":0.31384607157836014,"score_spread":0.22926279012833833,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417002434","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014129705,0.00014018829,0.9835001,0.00018217495,0.000023724553,0.00010459312,0.00007447609,0.00083064917,0.0010142871],"genre_scores_gemma":[0.62420595,0.00022475213,0.36992675,0.0004036686,0.0001351887,0.00032621346,0.00088586955,0.0003642987,0.0035273114],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987925,0.00032115346,0.000035410245,0.0004251828,0.00029710567,0.00012863],"domain_scores_gemma":[0.9978962,0.001037012,0.00017033728,0.0003632206,0.00041268318,0.00012059121],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016216417,0.0009782787,0.0011599284,0.0013370796,0.0009022691,0.0009577469,0.0028148205,0.0015871669,0.0021373192],"category_scores_gemma":[0.005344366,0.00046987075,0.0008784114,0.0009611593,0.0010802431,0.0026341043,0.0018649305,0.0017787421,0.00061327196],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018832035,0.00019737179,0.0032599922,0.00013946803,0.000097991106,0.00026799986,0.0003846461,0.66856015,0.010084058,0.026590452,0.006297856,0.28393167],"study_design_scores_gemma":[0.000011553783,0.000027913999,0.00015083129,0.0000035298933,0.000008894079,0.000040726878,0.000025801337,0.985908,0.0018636524,0.011150182,0.00080262165,0.0000062844215],"about_ca_topic_score_codex":0.0046316874,"about_ca_topic_score_gemma":0.006559136,"teacher_disagreement_score":0.0046316874,"about_ca_system_score_codex":0.0011166693,"about_ca_system_score_gemma":0.0012606801,"threshold_uncertainty_score":0.009209454},"labels":[],"label_agreement":null},{"id":"W4417036181","doi":"10.48550/arxiv.2512.04072","title":"SkillFactory: Self-Distillation For Learning Cognitive Behaviors","year":2025,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Alfred P. Sloan Foundation; Institute for Information and Communications Technology Promotion; Open Philanthropy Project; National Science Foundation","keywords":"Leverage (statistics); Reinforcement learning; Cognition; Initialization; Task (project management); Cognitive model; Priming (agriculture)","score_opus":0.059094410595447955,"score_gpt":0.23556282801553996,"score_spread":0.176468417420092,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417036181","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031441275,0.00032603816,0.9392928,0.00048632108,0.0001370045,0.00016367857,0.00048699477,0.02467614,0.0029897639],"genre_scores_gemma":[0.492708,0.0002332792,0.49654832,0.0008468965,0.00007468815,0.0005251569,0.0014142105,0.0017289907,0.0059203845],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994634,0.0001531979,0.00003112671,0.0001791726,0.00011133762,0.000061937346],"domain_scores_gemma":[0.9983719,0.0008922056,0.00012256991,0.0003988822,0.00012909125,0.00008534087],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015257514,0.0016354175,0.00075998803,0.00046780382,0.0003869907,0.0010718298,0.0029499133,0.001445245,0.006945706],"category_scores_gemma":[0.00634076,0.0009850455,0.0013218782,0.00034185383,0.0013184344,0.0024749269,0.0024187614,0.0033277979,0.001910838],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004902826,0.00033673726,0.002948282,0.00039306245,0.00019435941,0.00017174582,0.0002392622,0.6530398,0.011303304,0.024781268,0.010500509,0.29560137],"study_design_scores_gemma":[0.00002284895,0.000040630697,0.00006515115,0.000011085708,0.00000740456,0.000011355029,0.000004963948,0.98548764,0.00217199,0.011253864,0.00091509835,0.0000079095835],"about_ca_topic_score_codex":0.0040578744,"about_ca_topic_score_gemma":0.00768312,"teacher_disagreement_score":0.006945706,"about_ca_system_score_codex":0.0009323588,"about_ca_system_score_gemma":0.0014157614,"threshold_uncertainty_score":0.023235679},"labels":[],"label_agreement":null},{"id":"W4417068938","doi":"10.1109/iccv51701.2025.02126","title":"Aurelia: Test-Time Reasoning Distillation in Audio-Visual LLMs","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology; University of Toronto","funders":"","keywords":"Benchmark (surveying); Process (computing); Case-based reasoning; Visual reasoning; Reasoning system; Deductive reasoning; Code (set theory)","score_opus":0.006768323365754613,"score_gpt":0.2952767481250392,"score_spread":0.28850842475928457,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417068938","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09165498,0.0026613027,0.7215727,0.0021099248,0.0010088478,0.0008733981,0.0074017704,0.14779913,0.02491796],"genre_scores_gemma":[0.51577085,0.00031458796,0.45773762,0.0014731651,0.00012954991,0.0006677603,0.011450997,0.0037363486,0.0087190885],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975055,0.0008892804,0.00014044705,0.00076487765,0.00046507572,0.00023484348],"domain_scores_gemma":[0.99454147,0.0039319587,0.00016123767,0.00063072424,0.0005041535,0.00023056811],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028977026,0.00232258,0.00093651435,0.000813422,0.00057453697,0.002346424,0.0042274604,0.0028546997,0.018246135],"category_scores_gemma":[0.017984543,0.0006528738,0.0014243502,0.00046622707,0.0012672788,0.0032353094,0.0031264918,0.0035692998,0.0041251946],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017112065,0.00069324527,0.003892013,0.0016355418,0.0003139042,0.00071879197,0.0005042524,0.45508108,0.018542012,0.019346675,0.07020345,0.42735776],"study_design_scores_gemma":[0.00019564539,0.00010587374,0.00027199174,0.000048550995,0.000018728826,0.00006083321,0.00007758561,0.9748489,0.006860089,0.010234632,0.007255022,0.000022090351],"about_ca_topic_score_codex":0.013558992,"about_ca_topic_score_gemma":0.019846957,"teacher_disagreement_score":0.018246135,"about_ca_system_score_codex":0.0019196485,"about_ca_system_score_gemma":0.002239556,"threshold_uncertainty_score":0.061039448},"labels":[],"label_agreement":null},{"id":"W4417249200","doi":"10.1109/ivcnz67716.2025.11281867","title":"Visual Question Answering Using Multimodal Data Augmentation for Hausa","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Hausa; Question answering; Multimodality; Baseline (sea); Multimodal therapy; Multimodal interaction","score_opus":0.06130410411759673,"score_gpt":0.44171603674840976,"score_spread":0.380411932630813,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417249200","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08846789,0.0026965437,0.81709486,0.0016384308,0.00039074625,0.0008606318,0.0075529497,0.06896952,0.0123284245],"genre_scores_gemma":[0.49080744,0.00046993222,0.47394952,0.0011867591,0.00021402224,0.001058204,0.023451775,0.0012379464,0.0076242983],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99724776,0.0012733848,0.00011346752,0.0008587556,0.00036410164,0.00014248896],"domain_scores_gemma":[0.9971276,0.0016554577,0.000087908775,0.0006803162,0.00033483142,0.000113870956],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030507168,0.0016574251,0.0013585181,0.0016296267,0.00088197796,0.0021482673,0.0024508014,0.0018037527,0.010863236],"category_scores_gemma":[0.008705342,0.00043748264,0.0016041542,0.00092806603,0.0010591504,0.0049388404,0.005089073,0.0023112865,0.0047497354],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009803675,0.00068691286,0.0026383845,0.0009806106,0.0003001886,0.0005319324,0.0011732478,0.055846237,0.05456873,0.016732803,0.050982542,0.81457806],"study_design_scores_gemma":[0.0001228586,0.00045708756,0.0024822168,0.00011223011,0.00010036946,0.0005067495,0.00084453856,0.85900694,0.0466523,0.041290388,0.048286658,0.0001376916],"about_ca_topic_score_codex":0.005966649,"about_ca_topic_score_gemma":0.006103949,"teacher_disagreement_score":0.010863236,"about_ca_system_score_codex":0.001447207,"about_ca_system_score_gemma":0.0012376179,"threshold_uncertainty_score":0.03634119},"labels":[],"label_agreement":null},{"id":"W4417276938","doi":"10.48550/arxiv.2511.17366","title":"METIS: Multi-Source Egocentric Training for Integrated Dexterous Vision-Language-Action Model","year":2025,"lang":"","type":"preprint","venue":"ArXiv.org","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Bottleneck; Robustness (evolution); Teleoperation; Robot; Generalization; Robotics; Action (physics); Leverage (statistics)","score_opus":0.11413634178860264,"score_gpt":0.38653272736077604,"score_spread":0.2723963855721734,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417276938","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02810376,0.0005574184,0.95557255,0.0004335846,0.00013483058,0.00012357772,0.0011818366,0.011404187,0.0024881829],"genre_scores_gemma":[0.6431006,0.00048095523,0.33267355,0.0010161111,0.00011931447,0.0006757975,0.010531971,0.0012930145,0.010108682],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996524,0.0000729575,0.000012896937,0.00016047565,0.000051685696,0.000049591206],"domain_scores_gemma":[0.9995938,0.00013601163,0.000044734825,0.00011092003,0.00006876621,0.000045816738],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00071068323,0.00174077,0.00094453397,0.00059188064,0.0004497446,0.00074081833,0.0029404562,0.0015304012,0.0038084758],"category_scores_gemma":[0.002027091,0.00081832765,0.001372244,0.0006488566,0.0007223939,0.0014498209,0.00225812,0.0027173131,0.0017844688],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002621243,0.00034152388,0.002771362,0.00027791897,0.00030507607,0.00021780423,0.00019749955,0.65174305,0.0137336245,0.0086325575,0.024167046,0.2973504],"study_design_scores_gemma":[0.000009781078,0.000044971508,0.00024615778,0.000011317924,0.000008764783,0.00002392495,0.000013531551,0.993718,0.0012622506,0.003505663,0.0011469375,0.000008713453],"about_ca_topic_score_codex":0.008117096,"about_ca_topic_score_gemma":0.014907254,"teacher_disagreement_score":0.008117096,"about_ca_system_score_codex":0.0009185471,"about_ca_system_score_gemma":0.0014220305,"threshold_uncertainty_score":0.016139746},"labels":[],"label_agreement":null},{"id":"W4417283646","doi":"10.1145/3748636.3762801","title":"VisLN: An Interactive Visualization System for Vision Language Navigation","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary; University of Waterloo","funders":"","keywords":"Interpretability; Visualization; Debugging; Limiting; Field (mathematics); Interface (matter); User interface; Semantics (computer science)","score_opus":0.008161125548401855,"score_gpt":0.37837080423029906,"score_spread":0.3702096786818972,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417283646","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011424618,0.00037537963,0.7191882,0.0004460471,0.00019042061,0.00022086398,0.003978681,0.25154898,0.012626785],"genre_scores_gemma":[0.23443945,0.00072311686,0.7098407,0.0009060104,0.00009977627,0.0011249457,0.010635157,0.020914642,0.021316184],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9996792,0.00009541551,0.00002289245,0.00007304593,0.00009751506,0.000032039294],"domain_scores_gemma":[0.99924576,0.00037900833,0.00004749017,0.00011096266,0.00012396488,0.00009278689],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006738709,0.0012755255,0.0005171167,0.0007763647,0.00038230448,0.0012316861,0.0017302072,0.0010568263,0.031845614],"category_scores_gemma":[0.0026416304,0.00054608774,0.0006297001,0.00026293957,0.00042320904,0.0016754535,0.0026019388,0.0012630341,0.005542592],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024929498,0.0005095353,0.0060983626,0.0012514328,0.00022914531,0.0012211732,0.0029389819,0.030620664,0.108367346,0.032307193,0.30922678,0.5047365],"study_design_scores_gemma":[0.0005001749,0.00048573987,0.0025223468,0.0003096527,0.000115024974,0.00084632257,0.00035930824,0.4165361,0.062043544,0.029582456,0.48643282,0.00026652677],"about_ca_topic_score_codex":0.0025510653,"about_ca_topic_score_gemma":0.003958719,"teacher_disagreement_score":0.031845614,"about_ca_system_score_codex":0.0005472622,"about_ca_system_score_gemma":0.0008693409,"threshold_uncertainty_score":0.10653412},"labels":[],"label_agreement":null},{"id":"W4417283698","doi":"10.1145/3748636.3760462","title":"Cognitive Foundation Agents for Generalizable Vision-and-Language Navigation","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Embodied cognition; Task (project management); Cognition; Process (computing); Adaptation (eye); Asynchronous communication; Key (lock); Suite","score_opus":0.018163979812692393,"score_gpt":0.3740435197282998,"score_spread":0.3558795399156074,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417283698","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10838193,0.0004967453,0.8534271,0.0009567969,0.00013739863,0.0005073146,0.00032892922,0.002910484,0.032853283],"genre_scores_gemma":[0.6387324,0.0002154628,0.35666913,0.00015284531,0.000013465102,0.00040103548,0.00037660197,0.00018220654,0.003256879],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99836713,0.0007198511,0.00009433823,0.00019938844,0.00046749852,0.00015173186],"domain_scores_gemma":[0.9954725,0.0021055164,0.0003454356,0.0009867919,0.00080557127,0.0002842324],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031840722,0.0006599535,0.0003575631,0.00045332545,0.0010287213,0.0027482533,0.0012343646,0.0012251707,0.0050823228],"category_scores_gemma":[0.013130548,0.00033636467,0.00070569775,0.00027382243,0.0017818022,0.004162918,0.0031126502,0.0018394794,0.0005242392],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00073164556,0.000456583,0.0071344497,0.0007243771,0.00019305994,0.00028173573,0.0035949375,0.14033374,0.018120835,0.5874698,0.008752798,0.23220597],"study_design_scores_gemma":[0.00019530539,0.00051955227,0.002265125,0.00019925642,0.00014691433,0.00024682912,0.0010369371,0.70848763,0.014175525,0.22722866,0.04540748,0.0000908762],"about_ca_topic_score_codex":0.01486392,"about_ca_topic_score_gemma":0.012440394,"teacher_disagreement_score":0.01486392,"about_ca_system_score_codex":0.0015388048,"about_ca_system_score_gemma":0.0034118367,"threshold_uncertainty_score":0.029554784},"labels":[],"label_agreement":null},{"id":"W4417283734","doi":"10.1145/3748636.3762753","title":"SpatialGPT: Zero-Shot Vision-and-Language Navigation via Spatial CoT over Structured Spatial Memory","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates","keywords":"Leverage (statistics); Landmark; Spatial intelligence; Inference; Spatial contextual awareness; Task (project management); Generalization; Natural language; Context (archaeology)","score_opus":0.0059991082141304065,"score_gpt":0.2956740214560306,"score_spread":0.2896749132419002,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417283734","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018949395,0.00050767,0.94740885,0.00029191762,0.00018679716,0.00019544145,0.0008421619,0.025636505,0.005981181],"genre_scores_gemma":[0.39473164,0.00041955005,0.5893808,0.00069046946,0.00007035481,0.00039115257,0.0028576278,0.0010586118,0.0103998445],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99968505,0.000045293444,0.000014030264,0.00014268364,0.000072146606,0.000040877054],"domain_scores_gemma":[0.9996189,0.00011179877,0.000032120635,0.00012692623,0.000068047906,0.000042145984],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00048671145,0.0013181317,0.0007676211,0.0004073457,0.00052070705,0.0010253002,0.0034215301,0.0014196545,0.007214045],"category_scores_gemma":[0.0020465564,0.0005478102,0.0008553923,0.0004551805,0.0008513104,0.0029657716,0.0034032462,0.0019312267,0.0019163194],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004685975,0.00032553115,0.0019415446,0.00038871422,0.00020901518,0.00050377,0.0005060995,0.2435522,0.020568911,0.025083782,0.033817314,0.67263454],"study_design_scores_gemma":[0.0000454741,0.000112356014,0.00021481766,0.000024218507,0.00003200656,0.00010243164,0.000052039846,0.96681243,0.0053844755,0.021261439,0.005935787,0.000022434007],"about_ca_topic_score_codex":0.017909879,"about_ca_topic_score_gemma":0.02618777,"teacher_disagreement_score":0.017909879,"about_ca_system_score_codex":0.00093535596,"about_ca_system_score_gemma":0.0017168736,"threshold_uncertainty_score":0.03561127},"labels":[],"label_agreement":null},{"id":"W4417285899","doi":"10.1109/iccv51701.2025.01969","title":"VAMBA: Understanding Hour-Long Videos with Hybrid Mamba-Transformers","year":2025,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; University of Waterloo","funders":"","keywords":"ENCODE; Security token; Encoding (memory); Benchmark (surveying); Reduction (mathematics); Quadratic equation; Video tracking; Minification","score_opus":0.013961524523885186,"score_gpt":0.25840242052327045,"score_spread":0.24444089599938526,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417285899","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024802005,0.0009928847,0.963944,0.00036960165,0.00011750179,0.0000944465,0.00047867245,0.0067583644,0.002442382],"genre_scores_gemma":[0.56564873,0.0007135369,0.4193042,0.00069761573,0.00008411976,0.000314617,0.0018767597,0.0007009475,0.010659516],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99986947,0.000028816783,0.0000055471346,0.000049292295,0.00002583366,0.00002094729],"domain_scores_gemma":[0.99970454,0.00015349493,0.000019498764,0.000040204966,0.00005564235,0.00002655828],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00039864838,0.0010208086,0.000669008,0.00038694975,0.0002806607,0.00072461856,0.0018510531,0.001033943,0.0050352775],"category_scores_gemma":[0.0016181977,0.000421051,0.0008041713,0.00035358148,0.00033421358,0.0014632626,0.000984203,0.0017010642,0.0014292975],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044180706,0.00016120508,0.0015028224,0.00023722838,0.00022804155,0.00016215967,0.00014118901,0.54505014,0.02667367,0.007881176,0.009682993,0.40783763],"study_design_scores_gemma":[0.0000073769834,0.000024520512,0.000102962855,0.0000047861013,0.000005912746,0.000016980763,0.000014458536,0.9951923,0.0017989608,0.0020854233,0.0007416041,0.0000046739965],"about_ca_topic_score_codex":0.01256941,"about_ca_topic_score_gemma":0.020078031,"teacher_disagreement_score":0.01256941,"about_ca_system_score_codex":0.00070560933,"about_ca_system_score_gemma":0.0009476958,"threshold_uncertainty_score":0.024992466},"labels":[],"label_agreement":null},{"id":"W4417298816","doi":"10.48550/arxiv.2505.14685","title":"Language Models use Lookbacks to Track Beliefs","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Council for Higher Education; Azrieli Foundation; European Commission; Open Philanthropy Project; National Science Foundation","keywords":"Character (mathematics); Construct (python library); Visibility; Recall; Encoding (memory); State (computer science); Language model; Resolution (logic); Relation (database)","score_opus":0.050421206238252234,"score_gpt":0.32180394722060623,"score_spread":0.271382740982354,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417298816","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12607391,0.0005511878,0.85921097,0.0017414384,0.00006222568,0.00012663797,0.0018927441,0.0051877666,0.0051530995],"genre_scores_gemma":[0.8247536,0.00026745567,0.16985488,0.00036657753,0.000034070694,0.00013218468,0.0023251285,0.0005391232,0.0017269703],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9964059,0.0014463018,0.00024132172,0.0010509599,0.0005877995,0.00026755934],"domain_scores_gemma":[0.9796053,0.012645554,0.0022716685,0.0035450927,0.0014588218,0.00047354054],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037626598,0.0012110844,0.00088106445,0.0023365861,0.0009747252,0.0061912322,0.0029230455,0.001992395,0.0047006],"category_scores_gemma":[0.043598134,0.0014294384,0.002016026,0.0016712578,0.002556987,0.01422361,0.0034597078,0.0038464316,0.0012119134],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001236179,0.0003246788,0.07018127,0.0011494562,0.00090293336,0.0010996896,0.0141746225,0.27501592,0.019023733,0.306052,0.009050015,0.30178955],"study_design_scores_gemma":[0.000055821925,0.000152662,0.004304831,0.00012119514,0.000193794,0.00021958875,0.0011980825,0.6594239,0.006881164,0.31944007,0.007881615,0.00012721252],"about_ca_topic_score_codex":0.013737681,"about_ca_topic_score_gemma":0.0134494975,"teacher_disagreement_score":0.013737681,"about_ca_system_score_codex":0.0024565859,"about_ca_system_score_gemma":0.0015385507,"threshold_uncertainty_score":0.027315438},"labels":[],"label_agreement":null},{"id":"W4417322872","doi":"10.48550/arxiv.2510.21059","title":"Dynamic Retriever for In-Context Knowledge Editing via Policy Optimization","year":2025,"lang":"","type":"preprint","venue":"ArXiv.org","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Scalability; Task (project management); Software; Labrador Retriever; Source code; Collaborative editing","score_opus":0.026394916922155075,"score_gpt":0.33343174393782127,"score_spread":0.3070368270156662,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417322872","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025576966,0.00078939367,0.8719388,0.00077414716,0.00027236043,0.00024980184,0.000809257,0.09301225,0.006576932],"genre_scores_gemma":[0.45559725,0.0004267587,0.5155815,0.0011679346,0.00021937226,0.00047278331,0.0020363198,0.007962978,0.016535174],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985292,0.00035267015,0.00009970388,0.00048124278,0.00034607158,0.00019111359],"domain_scores_gemma":[0.9963007,0.0018966201,0.00019732375,0.0010100409,0.0003711536,0.00022417233],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021185675,0.0018596267,0.0016795908,0.0008629276,0.00059827353,0.0019815585,0.0036634926,0.002345336,0.017659413],"category_scores_gemma":[0.012604939,0.000713439,0.0010669047,0.0005488822,0.0009067468,0.0036471214,0.0034465415,0.0030724325,0.0070378557],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009870842,0.00071281055,0.0022651367,0.0007652846,0.00015674342,0.00053766114,0.0005417778,0.20088862,0.030291041,0.021098139,0.051061176,0.6906944],"study_design_scores_gemma":[0.000078266086,0.00009752446,0.00017507692,0.000024354762,0.00003160406,0.000115396746,0.000067964385,0.96675974,0.011235541,0.014205804,0.007174358,0.000034327542],"about_ca_topic_score_codex":0.00379167,"about_ca_topic_score_gemma":0.008681766,"teacher_disagreement_score":0.017659413,"about_ca_system_score_codex":0.0010593326,"about_ca_system_score_gemma":0.0018360338,"threshold_uncertainty_score":0.059076667},"labels":[],"label_agreement":null},{"id":"W4417423761","doi":"10.1109/iccv51701.2025.00061","title":"NegRefine: Refining Negative Label-Based Zero-Shot OOD Detection","year":2025,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada; Simon Fraser University","funders":"","keywords":"Subcategory; Set (abstract data type); Lexicon; Matching (statistics); Face (sociological concept); Word (group theory); Noun; Function (biology)","score_opus":0.02047313720632901,"score_gpt":0.3049961903917769,"score_spread":0.2845230531854479,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417423761","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07780358,0.0016504376,0.89309764,0.00035763235,0.000372338,0.00037112294,0.0015627898,0.018492034,0.0062924363],"genre_scores_gemma":[0.31156158,0.0006545458,0.6651684,0.0010140985,0.00020617481,0.00040297734,0.007405618,0.003175352,0.010411301],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9972953,0.00031417081,0.00014317289,0.000875871,0.001055189,0.00031619615],"domain_scores_gemma":[0.99563956,0.0013678526,0.00035606624,0.0008497135,0.0015973449,0.00018947614],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002335406,0.0018307823,0.001981948,0.0039480496,0.00097139797,0.002518448,0.0032609287,0.001861715,0.0035221244],"category_scores_gemma":[0.009118112,0.0006030754,0.001375194,0.0014850313,0.0013195266,0.0034268233,0.0043087327,0.0018561365,0.0023232466],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011148712,0.0004009099,0.010750219,0.0010582296,0.00022487581,0.00071526814,0.0006021605,0.016032632,0.09109939,0.006934052,0.02911853,0.8419489],"study_design_scores_gemma":[0.00011930298,0.00037114017,0.008018566,0.00017066106,0.00016727753,0.0015380544,0.0005288374,0.8234298,0.1137589,0.01809801,0.033657905,0.00014160002],"about_ca_topic_score_codex":0.0065364316,"about_ca_topic_score_gemma":0.0146154845,"teacher_disagreement_score":0.0065364316,"about_ca_system_score_codex":0.0010062342,"about_ca_system_score_gemma":0.0015507207,"threshold_uncertainty_score":0.012996733},"labels":[],"label_agreement":null},{"id":"W4417455182","doi":"10.1038/s41597-025-06365-y","title":"An Egocentric Life-Saving Interventional Procedure Dataset of Actions, Medical Questions, Maneuvers and Tools","year":2025,"lang":"en","type":"article","venue":"Scientific Data","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"U.S. Army Research Institute of Environmental Medicine; U.S. Army Medical Research and Development Command; Purdue University; School of Medicine, Indiana University; U.S. Department of Defense; National Science Foundation","keywords":"CLIPS; Action (physics); Resource (disambiguation); Object (grammar); Field (mathematics); Metadata","score_opus":0.04955760178494154,"score_gpt":0.3772135807482431,"score_spread":0.32765597896330156,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417455182","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027201299,0.0021875845,0.004446418,0.0006174945,0.00033431663,0.00036850662,0.9546243,0.0037434257,0.0064767697],"genre_scores_gemma":[0.017270865,0.00038472374,0.0055878726,0.0001873796,0.000049451708,0.00025597936,0.974502,0.000105376304,0.0016563917],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99917114,0.00014269477,0.00007715834,0.00025851207,0.00023844541,0.00011208971],"domain_scores_gemma":[0.99883884,0.00031697043,0.00012583467,0.00023563355,0.00029893132,0.00018370053],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00055339123,0.001918769,0.000883133,0.0024453844,0.00069087354,0.0009029025,0.0016631301,0.002200801,0.007632561],"category_scores_gemma":[0.0024886476,0.00029729874,0.0011292993,0.002290029,0.0005106703,0.0007519324,0.0013473514,0.0012199437,0.009349943],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009992669,0.00078173936,0.015544033,0.0029844167,0.00026563095,0.0010216661,0.00043058465,0.0061230944,0.0063045854,0.0014696091,0.87620044,0.08787504],"study_design_scores_gemma":[0.00039460356,0.00054525444,0.08773055,0.0011456672,0.0002299515,0.002137885,0.0014486997,0.026091695,0.009688511,0.0036000751,0.8667606,0.00022653304],"about_ca_topic_score_codex":0.0229802,"about_ca_topic_score_gemma":0.07182089,"teacher_disagreement_score":0.0229802,"about_ca_system_score_codex":0.0012325095,"about_ca_system_score_gemma":0.0014783744,"threshold_uncertainty_score":0.04569292},"labels":[],"label_agreement":null},{"id":"W4417515405","doi":"10.1109/tvt.2025.3608811","title":"DriveSOTIF: Advancing SOTIF Through Multimodal Large Language Models","year":2025,"lang":"en","type":"preprint","venue":"IEEE Transactions on Vehicular Technology","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of New Brunswick; University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Benchmarking; Inference; Baseline (sea); Domain (mathematical analysis); Work (physics); Language model; Task analysis; Language understanding","score_opus":0.009577541560101646,"score_gpt":0.28380894074969765,"score_spread":0.274231399189596,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417515405","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16007702,0.0041457475,0.6613215,0.0029353437,0.0011140787,0.00093621673,0.0408315,0.11297246,0.01566609],"genre_scores_gemma":[0.48895428,0.0009743486,0.3820063,0.0016792205,0.00022494316,0.0009064034,0.109700054,0.00302741,0.01252719],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99907887,0.00028609252,0.000051530817,0.00034192397,0.00013915576,0.00010245274],"domain_scores_gemma":[0.99867487,0.0006225393,0.00005882007,0.00029074415,0.00027598575,0.000077047756],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015075734,0.0021871638,0.00074873224,0.0012798517,0.0005517348,0.0017383496,0.0022296708,0.00187481,0.004676202],"category_scores_gemma":[0.0053238156,0.0005072177,0.002056699,0.0006551388,0.0005196175,0.0025778618,0.0021567969,0.0027819185,0.0035739103],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005955926,0.0007735951,0.0112271,0.00089445966,0.0005445663,0.000592357,0.0005880696,0.3262733,0.018103428,0.006576207,0.14274,0.49109134],"study_design_scores_gemma":[0.00006657388,0.00010426323,0.0013814834,0.00006753371,0.000049240927,0.00011918663,0.00020685328,0.9685298,0.005766645,0.0070410576,0.016611747,0.00005569377],"about_ca_topic_score_codex":0.03244457,"about_ca_topic_score_gemma":0.049208526,"teacher_disagreement_score":0.03244457,"about_ca_system_score_codex":0.0014778444,"about_ca_system_score_gemma":0.0015725167,"threshold_uncertainty_score":0.06451142},"labels":[],"label_agreement":null},{"id":"W4417517804","doi":"10.48550/arxiv.2505.08455","title":"VCRBench: Exploring Long-form Causal Reasoning Capabilities of Large Video Language Models","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Mitacs","keywords":"Causal reasoning; Causal model; Visual reasoning; Key (lock); Benchmark (surveying); Language model; Simple (philosophy); Modular design; Model-based reasoning","score_opus":0.05864879819581904,"score_gpt":0.3154730716831285,"score_spread":0.25682427348730946,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417517804","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09501944,0.0033036608,0.83701885,0.001972229,0.00038530133,0.00067674613,0.008384991,0.044380225,0.008858585],"genre_scores_gemma":[0.5736766,0.00092340494,0.4021665,0.0010928456,0.00010616681,0.00053541287,0.016421458,0.0017636683,0.0033140383],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99731344,0.0011307071,0.00015092912,0.0007577697,0.0004891857,0.00015800772],"domain_scores_gemma":[0.98763347,0.009863315,0.00042552242,0.001013416,0.0007749196,0.000289418],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037708413,0.002054026,0.0008464076,0.0013665591,0.0005876452,0.002558055,0.003482931,0.0019354987,0.0071130926],"category_scores_gemma":[0.024642639,0.0005815263,0.0017735182,0.0007300691,0.0010264125,0.0049135825,0.0022855015,0.0032917808,0.0018624876],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009647295,0.0005808013,0.0052952836,0.0021491114,0.0003690981,0.0006962839,0.00091216766,0.55713373,0.0133368885,0.032687895,0.029927036,0.35594696],"study_design_scores_gemma":[0.00006000471,0.000095784344,0.00031390667,0.000055282253,0.000022927003,0.000060954586,0.00011810171,0.9723218,0.0035783108,0.020188196,0.0031597535,0.000024947907],"about_ca_topic_score_codex":0.025546148,"about_ca_topic_score_gemma":0.029574377,"teacher_disagreement_score":0.025546148,"about_ca_system_score_codex":0.0024204531,"about_ca_system_score_gemma":0.0023777063,"threshold_uncertainty_score":0.0507949},"labels":[],"label_agreement":null},{"id":"W6888591465","doi":"10.2021/ju.v2i1.2493","title":"Songs of Prescience: Canadian Musical Activism in Climate Breakdown","year":2022,"lang":"en","type":"article","venue":"The Journal of Macrodynamic Analysis (Memorial University of Newfoundland)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Musical; Climate change; Agency (philosophy); Government (linguistics)","score_opus":0.00571727521634234,"score_gpt":0.2124514039836661,"score_spread":0.20673412876732378,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6888591465","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6616986,0.0019014318,0.0005424336,0.016584296,0.00048641377,0.000056351986,0.0009105633,0.00004645651,0.31777358],"genre_scores_gemma":[0.98650837,0.0003994998,0.00009730568,0.0005436048,0.00003465447,0.000010214304,0.00010775853,0.000021474154,0.012277047],"study_design_codex":"qualitative","study_design_gemma":"not_applicable","domain_scores_codex":[0.99877995,0.00014198718,0.00001805526,0.00012418166,0.00037381853,0.00056192087],"domain_scores_gemma":[0.9973563,0.0005369043,0.00018153983,0.00008466905,0.000744188,0.001096471],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001157949,0.0003300838,0.00025758735,0.0019864258,0.01566714,0.00604829,0.0011508041,0.0013932077,0.014411341],"category_scores_gemma":[0.0062264535,0.00020540271,0.00012903579,0.0040956405,0.0054743867,0.0012186052,0.0037274167,0.0023646296,0.00043228292],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00086179236,0.00014183907,0.07989806,0.00031217575,0.00008674302,0.0016362478,0.5190879,0.00092690875,0.004864006,0.15128118,0.109161116,0.13174209],"study_design_scores_gemma":[0.0000279042,0.00003993621,0.20598398,0.00013023168,0.000026246591,0.00015467092,0.4260373,0.0003524565,0.00051042234,0.0032550753,0.36340594,0.000075859],"about_ca_topic_score_codex":0.96296203,"about_ca_topic_score_gemma":0.9883477,"teacher_disagreement_score":0.04150758,"about_ca_system_score_codex":0.04150758,"about_ca_system_score_gemma":0.033726823,"threshold_uncertainty_score":0.30115998},"labels":[],"label_agreement":null},{"id":"W6889145698","doi":"10.25384/sage.22097725","title":"Supplemental Material - Sex-specific frailty and chronological age normative carotid artery intima-media thickness values using the Canadian longitudinal study of aging","year":2023,"lang":"en","type":"article","venue":"Sage Journals Data","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Normative; Longitudinal study; Carotid arteries; Ageing; Common carotid artery; Longitudinal data; Ultrasound; Vascular disease","score_opus":0.14693944079663723,"score_gpt":0.3703898591655418,"score_spread":0.22345041836890456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6889145698","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0030846344,0.00016690393,0.0009809415,0.0003982351,0.0002478915,0.00034537286,0.99068284,0.00044022276,0.003653038],"genre_scores_gemma":[0.021890415,0.0007700867,0.01068435,0.0009744561,0.00033811238,0.0018433789,0.9429324,0.0005470073,0.020019762],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.9990759,0.00010527233,0.00014781303,0.0001236582,0.0004159617,0.00013135908],"domain_scores_gemma":[0.98381203,0.005135088,0.00089720637,0.00096069975,0.008354474,0.00084046787],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0015309425,0.0009846288,0.0010784977,0.0042896373,0.0015805078,0.0013285537,0.0023368686,0.0011131618,0.45897982],"category_scores_gemma":[0.024638444,0.00072755764,0.0010036516,0.005237058,0.00020532108,0.0007607355,0.0008254156,0.0009575275,0.05030051],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034874261,0.00042017532,0.024098113,0.00084956264,0.0001115715,0.0001300106,0.0001131201,0.00024314022,0.00027221921,0.0005431215,0.9474264,0.025443785],"study_design_scores_gemma":[0.0014520636,0.00037776562,0.49058497,0.0021751341,0.0004192343,0.0020641547,0.0008889561,0.0016340617,0.0014629073,0.0042609465,0.49441525,0.00026451267],"about_ca_topic_score_codex":0.23069279,"about_ca_topic_score_gemma":0.34549457,"teacher_disagreement_score":0.7693072,"about_ca_system_score_codex":0.0022233669,"about_ca_system_score_gemma":0.0050988845,"threshold_uncertainty_score":0.7716996},"labels":[],"label_agreement":null},{"id":"W6892225984","doi":"10.5075/epfl-thesis-10642","title":"Infusing structured knowledge priors in neural models for sample-efficient symbolic reasoning","year":2024,"lang":"en","type":"dissertation","venue":"Infoscience (Ecole Polytechnique Fédérale de Lausanne)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Instituto de Ciencias del Mar y Limnología, Universidad Nacional Autónoma de México; Institute for Catastrophic Loss Reduction","keywords":"Generalization; Artificial neural network; Prior probability; Representation (politics); Commonsense knowledge; Plan (archaeology); Knowledge representation and reasoning; Commonsense reasoning; Qualitative reasoning","score_opus":0.014842905866114312,"score_gpt":0.31280392241144955,"score_spread":0.29796101654533524,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6892225984","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.057301663,0.00021507136,0.93820477,0.00088811415,0.000034222874,0.000057977988,0.00011341942,0.00054413965,0.0026406792],"genre_scores_gemma":[0.77322036,0.00033372687,0.22315753,0.00024542087,0.000054881624,0.00018697347,0.00025283496,0.00013432791,0.0024139364],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9991622,0.00038088337,0.000047367743,0.00016759422,0.00015752173,0.00008451698],"domain_scores_gemma":[0.9916338,0.0063268268,0.0005941852,0.00083604443,0.00039767363,0.00021152488],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002344427,0.0008309385,0.00097365276,0.00066544034,0.0004907119,0.0020710514,0.0019461901,0.0016031106,0.0024681415],"category_scores_gemma":[0.0151705425,0.0008079993,0.0010843539,0.00062364683,0.0019495701,0.004238042,0.002237379,0.00319659,0.0004440226],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000082796374,0.000048238766,0.0006590268,0.00005453006,0.00003614702,0.000042717336,0.00014173174,0.9345385,0.001043431,0.044322483,0.00040696398,0.01862349],"study_design_scores_gemma":[0.0000061574005,0.000008931861,0.00003723274,0.0000062391623,0.000004711139,0.0000042032734,0.000005443358,0.97637343,0.00026949198,0.023159679,0.00012083489,0.0000034840111],"about_ca_topic_score_codex":0.0044376543,"about_ca_topic_score_gemma":0.0076788603,"teacher_disagreement_score":0.0044376543,"about_ca_system_score_codex":0.0019458354,"about_ca_system_score_gemma":0.0013638884,"threshold_uncertainty_score":0.014118135},"labels":[],"label_agreement":null},{"id":"W6893799380","doi":"10.5281/zenodo.4804705","title":"[Download~^Mp3] Georgia Anne Muldrow - Overload Album Download","year":2021,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Download; Information overload; Reading (process)","score_opus":0.02121420118311327,"score_gpt":0.25456332880264326,"score_spread":0.23334912761953,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6893799380","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005250902,0.00056036946,0.0106283175,0.0011673284,0.0018108607,0.00029192647,0.057408158,0.21745303,0.71015483],"genre_scores_gemma":[0.0030296405,0.00046249814,0.0048802514,0.0010733487,0.0003839788,0.00033522194,0.043175075,0.11093957,0.8357205],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996531,0.000027404049,0.000015192765,0.0000742947,0.00017435601,0.00005575555],"domain_scores_gemma":[0.99839276,0.00021316658,0.000049116483,0.00031184492,0.0007066062,0.00032645153],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0004834314,0.002004841,0.0009885785,0.0016862898,0.001329053,0.0041207406,0.0018690919,0.0013581191,0.8623853],"category_scores_gemma":[0.004247232,0.0009287738,0.00085124804,0.0017191815,0.0003510992,0.0051538795,0.004187651,0.0016354619,0.88849914],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000019053674,0.0000038948656,0.000023363044,0.000046571073,0.0000011808303,0.000014456931,0.000015509167,0.00001844051,0.00014831283,0.00028553364,0.98625493,0.013168643],"study_design_scores_gemma":[0.000011819825,0.0000070655633,0.00021900103,0.00004449174,0.0000028394484,0.00004606988,0.00001806152,0.0001478374,0.0004539313,0.0005835309,0.99845004,0.000015269901],"about_ca_topic_score_codex":0.005354621,"about_ca_topic_score_gemma":0.007897763,"teacher_disagreement_score":0.13761473,"about_ca_system_score_codex":0.0010329678,"about_ca_system_score_gemma":0.00062848924,"threshold_uncertainty_score":0.19629061},"labels":[],"label_agreement":null},{"id":"W6901676980","doi":"10.60692/4sy44-gqy76","title":"Mind the Context: The Impact of Contextualization in Neural Module Networks for Grounding Visual Referring Expressions","year":2021,"lang":"en","type":"article","venue":"Greater South Information System","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Contextualization; Generalization; Set (abstract data type); Exploit; Parameterized complexity; Cube (algebra); Test set; Implementation","score_opus":0.0418449064862477,"score_gpt":0.2992807166771646,"score_spread":0.2574358101909169,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6901676980","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34488717,0.0040735505,0.6210233,0.0019684825,0.00033325836,0.00018598164,0.0009781845,0.012462241,0.014087887],"genre_scores_gemma":[0.882308,0.00056310056,0.11038811,0.0007483577,0.00011304859,0.0001387635,0.001403192,0.00044452705,0.003892906],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99931526,0.00025700303,0.000025720545,0.00028777297,0.000048246173,0.00006603645],"domain_scores_gemma":[0.99922013,0.000393972,0.00007566831,0.0001666888,0.000098380915,0.000045184384],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010788332,0.0014869269,0.00057892955,0.00038385327,0.00048575416,0.0011220783,0.0019974483,0.0013547597,0.0034073584],"category_scores_gemma":[0.0044407374,0.0005083755,0.0009924402,0.00047480655,0.00069609267,0.0036504483,0.0016880729,0.002028863,0.00092007074],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009555078,0.0003321617,0.008352417,0.00046030956,0.00035089912,0.0005309218,0.0007072735,0.42765597,0.038551174,0.02106779,0.013649513,0.487386],"study_design_scores_gemma":[0.000033627915,0.0001242596,0.0011898692,0.000051446954,0.00010313062,0.00010275706,0.00007210169,0.9757529,0.0063278093,0.011971145,0.0042454205,0.000025547804],"about_ca_topic_score_codex":0.0051611285,"about_ca_topic_score_gemma":0.011362897,"teacher_disagreement_score":0.0051611285,"about_ca_system_score_codex":0.0010249707,"about_ca_system_score_gemma":0.0006630484,"threshold_uncertainty_score":0.011398733},"labels":[],"label_agreement":null},{"id":"W6902603111","doi":"10.6084/m9.figshare.c.7906156","title":"Health and social service provider perspectives on challenges, approaches, and recommendations for treating long COVID: a qualitative study of Canadian provider experiences","year":2025,"lang":"en","type":"other","venue":"Figshare","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Toronto; University Health Network; Centre for Addiction and Mental Health","funders":"","keywords":"Mental health; Thematic analysis; Psychosocial; Qualitative research; Service provider; Psychoeducation; Health care; Intervention (counseling)","score_opus":0.16322238570063383,"score_gpt":0.3940439448848523,"score_spread":0.23082155918421846,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6902603111","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9728442,0.0018169652,0.0015921269,0.010743926,0.0001999425,0.00045653826,0.00045691134,0.000027783859,0.011861656],"genre_scores_gemma":[0.99128425,0.0015663047,0.0011581767,0.0024137287,0.00003162879,0.00020780494,0.0001254911,0.00004442756,0.0031681834],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.98310167,0.008551869,0.00051017595,0.0008477334,0.0018860388,0.0051025674],"domain_scores_gemma":[0.97163796,0.015245521,0.0017592434,0.0004991244,0.0050134496,0.0058448324],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014557082,0.0007847339,0.0010009444,0.002250758,0.028411409,0.007298327,0.0031569416,0.0025316048,0.003921918],"category_scores_gemma":[0.027128944,0.0010267553,0.0006296047,0.004521844,0.015661687,0.0034147566,0.0072310893,0.0054791183,0.00023609056],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000020196378,0.000016014359,0.0016355901,0.00008606424,0.0000030588903,0.0005096532,0.9943568,0.000030964693,0.00018319809,0.0007006054,0.0007547732,0.0017029186],"study_design_scores_gemma":[0.0000019854197,0.000008435655,0.00058675464,0.00009118097,0.0000020638718,0.00005972823,0.9948126,0.000035048713,0.000033301832,0.000053599866,0.0043057473,0.000009620912],"about_ca_topic_score_codex":0.8578601,"about_ca_topic_score_gemma":0.9055877,"teacher_disagreement_score":0.14213991,"about_ca_system_score_codex":0.06270931,"about_ca_system_score_gemma":0.10512357,"threshold_uncertainty_score":0.45499003},"labels":[],"label_agreement":null},{"id":"W6903066260","doi":"10.1016/j.neucom.2025.130979","title":"MKE-PLLM: A benchmark for multilingual knowledge editing on pretrained large language model","year":2025,"lang":"en","type":"article","venue":"Neurocomputing","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Major Science and Technology Projects in Yunnan Province; Applied Basic Research Foundation of Yunnan Province; Yunnan Provincial Science and Technology Department; National Natural Science Foundation of China","keywords":"Benchmark (surveying); Usability; Robustness (evolution); Focus (optics); Language model; Baseline (sea)","score_opus":0.013848097286401212,"score_gpt":0.3358910638878189,"score_spread":0.32204296660141774,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6903066260","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23053658,0.014657691,0.23971903,0.002801784,0.004436209,0.0018170978,0.10014601,0.3562827,0.049602997],"genre_scores_gemma":[0.30100235,0.002098854,0.39164725,0.0011445264,0.00032984983,0.0011900828,0.2737671,0.01142132,0.017398676],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99373394,0.0017979948,0.00077588484,0.0019103994,0.0011624846,0.0006192496],"domain_scores_gemma":[0.98370045,0.008326531,0.0003561938,0.0040950454,0.00277825,0.00074355386],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005858644,0.0041403845,0.0023490845,0.0047363127,0.002000519,0.0041322703,0.0076365853,0.004624597,0.025886485],"category_scores_gemma":[0.031496268,0.0014512913,0.002466347,0.0043365266,0.0010509528,0.00827892,0.0053707818,0.0044868723,0.015885096],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024474768,0.001711979,0.0029523557,0.0027883383,0.0013207769,0.0010276622,0.0003782278,0.07466412,0.007916692,0.0040717316,0.24010718,0.6606134],"study_design_scores_gemma":[0.0012684019,0.00096137304,0.004036514,0.00037440006,0.0005033383,0.0011614097,0.0009365422,0.8584063,0.041826583,0.01626198,0.07399677,0.00026635444],"about_ca_topic_score_codex":0.021441909,"about_ca_topic_score_gemma":0.0322669,"teacher_disagreement_score":0.025886485,"about_ca_system_score_codex":0.0021252402,"about_ca_system_score_gemma":0.0040121316,"threshold_uncertainty_score":0.08659887},"labels":[],"label_agreement":null},{"id":"W6906665089","doi":"10.17632/3kt8gyckfb.3","title":"Risk of violence in elderly people in Brazil: representativeness of the age group // Risco de violência em pessoas idosas no Brasil: representatividade da faixa etária","year":2025,"lang":"en","type":"dataset","venue":"Mendeley Data","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Representativeness heuristic; Depression (economics); Public health; Elderly people; Scale (ratio); Odds; Geriatric Depression Scale; Age groups","score_opus":0.022087678782424795,"score_gpt":0.34694101141329153,"score_spread":0.32485333263086674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6906665089","genre_codex":"empirical","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99738866,0.0006212858,0.00015805084,0.000105918414,0.0000057246775,0.00003054654,0.0002653736,0.0000044236876,0.0014200605],"genre_scores_gemma":[0.99923813,0.0003478462,0.00015428296,0.000027190754,0.0000040306977,0.000013093781,0.00010708221,0.0000013830995,0.00010693261],"study_design_codex":"observational","study_design_gemma":"not_applicable","domain_scores_codex":[0.9994636,0.00013028935,0.00008301963,0.00008671481,0.00014217777,0.000094202755],"domain_scores_gemma":[0.9990878,0.00017372082,0.00043668205,0.00010240598,0.00009774965,0.00010167148],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00085450197,0.00019691617,0.00036444445,0.0013972378,0.00043402892,0.0005637511,0.00028233082,0.00038447382,0.0007521696],"category_scores_gemma":[0.0027920257,0.00038103154,0.0005055529,0.0010322792,0.0003595614,0.00032253613,0.0006238507,0.00024065228,0.0001216343],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000012579328,0.000019717061,0.9959971,0.000029514478,0.000025454478,0.000049172526,0.00090389524,0.00001206798,0.00019061442,0.000052785574,0.00003476077,0.0026723272],"study_design_scores_gemma":[0.0000013702559,0.000044017303,0.9973609,0.00003898092,0.000023848746,0.000273364,0.0017663945,0.0000864236,0.00005302684,0.0000504517,0.0002972493,0.0000040124196],"about_ca_topic_score_codex":0.034737226,"about_ca_topic_score_gemma":0.07201961,"teacher_disagreement_score":0.034737226,"about_ca_system_score_codex":0.00038300562,"about_ca_system_score_gemma":0.0005329865,"threshold_uncertainty_score":0.0690701},"labels":[],"label_agreement":null},{"id":"W6920901177","doi":"10.60692/rz8hd-c0104","title":"Pragmatic Inference with a CLIP Listener for Contrastive Captioning","year":2023,"lang":"en","type":"article","venue":"Greater South Information System","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Closed captioning; Discriminative model; Inference; Leverage (statistics); Fluency; Prosody; Hyperparameter","score_opus":0.031017765884502436,"score_gpt":0.2535954053786056,"score_spread":0.22257763949410314,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6920901177","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0063938876,0.00017838237,0.9836436,0.00031213707,0.00011227198,0.00014685262,0.00019992153,0.004381841,0.0046311845],"genre_scores_gemma":[0.32987347,0.00018056334,0.65967995,0.0009103273,0.00031520973,0.00040946956,0.0012299732,0.0012914938,0.0061093844],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99670863,0.0014030369,0.00011987588,0.0009272486,0.0006637373,0.0001774432],"domain_scores_gemma":[0.99504894,0.0028910222,0.00029494383,0.00085087534,0.00069679477,0.0002174402],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035636213,0.0018368625,0.00093751604,0.0010641306,0.0009560479,0.0025904903,0.00270819,0.0020933154,0.010995838],"category_scores_gemma":[0.018255511,0.0008564656,0.001344991,0.0005509058,0.0017535733,0.0037981288,0.0029604395,0.003666472,0.0032117546],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00097539934,0.00036921864,0.0023361212,0.0008431317,0.00032648377,0.00072311016,0.0023481783,0.12288969,0.072695255,0.10901665,0.039415017,0.64806175],"study_design_scores_gemma":[0.000054853645,0.00010911707,0.00049178064,0.000039953182,0.000060974566,0.00020048066,0.00014503027,0.9189247,0.02080215,0.04785566,0.01124792,0.00006744843],"about_ca_topic_score_codex":0.0025570297,"about_ca_topic_score_gemma":0.0035055815,"teacher_disagreement_score":0.010995838,"about_ca_system_score_codex":0.0014743606,"about_ca_system_score_gemma":0.001260921,"threshold_uncertainty_score":0.03678471},"labels":[],"label_agreement":null},{"id":"W6923552786","doi":"10.14288/1.0428359","title":"Millsite, Cultus Lake","year":2023,"lang":"en","type":"other","venue":"cIRcle (University of British Columbia)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"FLAGS register; Period (music); Field (mathematics); Lake district","score_opus":0.007062395759626506,"score_gpt":0.18946458951139203,"score_spread":0.18240219375176553,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6923552786","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00374835,0.00022817834,0.0002833679,0.0021657394,0.0005428,0.000040991003,0.0037711652,0.0007230169,0.9884964],"genre_scores_gemma":[0.0076230424,0.00013933764,0.0003042221,0.0001863813,0.000027519518,0.000013324891,0.0009873407,0.00016941491,0.9905494],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9998876,0.0000051960687,0.0000014921725,0.00001884189,0.000044628607,0.000042276566],"domain_scores_gemma":[0.99983644,0.000009075695,0.0000049806076,0.000006241174,0.00006421509,0.00007897913],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00008478226,0.0004503499,0.00016540638,0.0005714848,0.0049042003,0.0026349088,0.00042404153,0.00079795433,0.5424017],"category_scores_gemma":[0.00030603353,0.00027595102,0.00014945571,0.0009579449,0.0006193075,0.0011048139,0.0011869039,0.0011758897,0.13979144],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000039433173,0.0000083722125,0.0005451167,0.000027776767,6.513545e-7,0.00014866819,0.00035041285,0.000043397144,0.00039491357,0.0037181014,0.9781807,0.016542405],"study_design_scores_gemma":[0.000004598212,0.0000050369204,0.002086439,0.000018373597,7.9082395e-7,0.000036856403,0.000717197,0.00005610082,0.00012225463,0.00018386474,0.9967643,0.0000041524113],"about_ca_topic_score_codex":0.32102284,"about_ca_topic_score_gemma":0.81418955,"teacher_disagreement_score":0.67897713,"about_ca_system_score_codex":0.003541058,"about_ca_system_score_gemma":0.0030171373,"threshold_uncertainty_score":0.6527085},"labels":[],"label_agreement":null},{"id":"W6950241977","doi":"10.5683/sp3/j5ebgi","title":"Enquête sur la population active, mars 2003 [Canada] [Remanié Recensement 2011]","year":2023,"lang":"fr","type":"dataset","venue":"Borealis","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Statistics Canada","funders":"","keywords":"Population; Research methodology; Sugar industry; Irrigation district","score_opus":0.020696347780857084,"score_gpt":0.27558383669010195,"score_spread":0.25488748890924484,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6950241977","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029471949,0.01385552,0.0011083987,0.011418944,0.0014878516,0.00029540795,0.8728215,0.00030546176,0.06923501],"genre_scores_gemma":[0.18863323,0.031047065,0.005576515,0.0085045,0.00062698947,0.00079695025,0.48543406,0.00033622544,0.2790445],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985677,0.00006298054,0.00011207192,0.00016981734,0.00078952784,0.00029789069],"domain_scores_gemma":[0.9931899,0.00020391408,0.00021756739,0.00010668486,0.0057887733,0.00049307296],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017197864,0.00058085547,0.0004980868,0.0048651635,0.0034672818,0.0028456931,0.001451046,0.00084690173,0.010892033],"category_scores_gemma":[0.0050587384,0.00046194522,0.000673092,0.013685731,0.00055545714,0.00066180876,0.0009515152,0.0014814298,0.0025148883],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013724295,0.000037541104,0.0710804,0.00074792,0.00011287908,0.00018423446,0.0027250545,0.00047894503,0.000206249,0.0034775524,0.8709729,0.04983919],"study_design_scores_gemma":[0.000029065211,0.000021586706,0.4112945,0.00062467763,0.00006189878,0.00006404236,0.002644924,0.00024188543,0.00032868978,0.00021151958,0.58441633,0.00006087094],"about_ca_topic_score_codex":0.99813133,"about_ca_topic_score_gemma":0.9983997,"teacher_disagreement_score":0.043694638,"about_ca_system_score_codex":0.043694638,"about_ca_system_score_gemma":0.094693825,"threshold_uncertainty_score":0.31702828},"labels":[],"label_agreement":null},{"id":"W6959225239","doi":"10.7488/ds/4481","title":"Tower Blocks UK: Glasgow City Red Road, glw6-16.jpg","year":2023,"lang":"en","type":"other","venue":"University of Edinburgh","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Tower; Block (permutation group theory); Corporation; Project commissioning; Quarter (Canadian coin); Architecture; Frame (networking)","score_opus":0.012415950960147632,"score_gpt":0.22580329123917134,"score_spread":0.2133873402790237,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6959225239","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0014963257,0.0014981846,0.0004962359,0.0024566876,0.00095717894,0.00017203772,0.020690113,0.0031547684,0.96907836],"genre_scores_gemma":[0.0027669196,0.0005869486,0.00018942983,0.00033112566,0.00007349109,0.000041201372,0.0038490824,0.00081659836,0.9913453],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99943286,0.000051156698,0.00002809587,0.0001610266,0.00020925156,0.00011760359],"domain_scores_gemma":[0.9988539,0.00012241876,0.0000770485,0.00016839855,0.00036600683,0.0004121443],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.00040878422,0.0015273642,0.0010037348,0.0014056902,0.0019225092,0.0067618014,0.0012588366,0.0034962196,0.95148784],"category_scores_gemma":[0.0018149172,0.00068818004,0.000444784,0.0030604838,0.0010796944,0.0039028986,0.0032679338,0.001614441,0.8876057],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000060647868,0.000010954928,0.00020498467,0.00018384102,0.0000028696109,0.0001855276,0.00013460132,0.000045702072,0.0002638894,0.001695647,0.9572778,0.039933566],"study_design_scores_gemma":[0.000014871474,0.000021148502,0.0012946737,0.000108234235,0.0000021604844,0.00007178093,0.00022954459,0.000018605431,0.00006291304,0.00020412418,0.9979645,0.000007429843],"about_ca_topic_score_codex":0.020397766,"about_ca_topic_score_gemma":0.05556668,"teacher_disagreement_score":0.04851216,"about_ca_system_score_codex":0.001502043,"about_ca_system_score_gemma":0.0014980414,"threshold_uncertainty_score":0.0691967},"labels":[],"label_agreement":null},{"id":"W7008927071","doi":"","title":"A deep learning approach for automatically generating descriptions of images containing people","year":2018,"lang":"en","type":"dissertation","venue":"Library Open Repository (Universidad Complutense Madrid)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Canadian Institute for Advanced Research","keywords":"Deep learning; Task (project management); Field (mathematics); Generator (circuit theory); Image (mathematics); Artificial neural network; Image processing","score_opus":0.0137681633948925,"score_gpt":0.25263805203592743,"score_spread":0.23886988864103492,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7008927071","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019998906,0.00111806,0.95975244,0.000881828,0.00020606272,0.000285085,0.002911712,0.009024761,0.0058212164],"genre_scores_gemma":[0.19432478,0.0009946668,0.7801675,0.0005577644,0.00008104036,0.0002811391,0.00886159,0.00043109243,0.01430043],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994535,0.00012860286,0.00004067849,0.00018764213,0.00013841063,0.00005121262],"domain_scores_gemma":[0.9989735,0.00043542573,0.000102107966,0.00018619794,0.00023444205,0.00006838182],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00080406026,0.0011385867,0.00045424164,0.0012305048,0.0004071743,0.00087783585,0.0018244784,0.0012363496,0.0050098943],"category_scores_gemma":[0.0026667363,0.0006346859,0.0013720753,0.0006729881,0.0006786605,0.0026130737,0.0011178082,0.002418116,0.0016806409],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000306041,0.0003228962,0.0011536047,0.00059412216,0.000118284785,0.00040711657,0.00039527635,0.0838815,0.01974058,0.026046641,0.035423066,0.8316108],"study_design_scores_gemma":[0.000036233698,0.00010973313,0.0005001667,0.00008604629,0.00004310111,0.00024246839,0.00013131295,0.931541,0.02439833,0.022588188,0.020291433,0.000032031523],"about_ca_topic_score_codex":0.008591491,"about_ca_topic_score_gemma":0.012380711,"teacher_disagreement_score":0.008591491,"about_ca_system_score_codex":0.0016754278,"about_ca_system_score_gemma":0.0013096469,"threshold_uncertainty_score":0.01708299},"labels":[],"label_agreement":null},{"id":"W7020581563","doi":"","title":"[Letter from P. H. Raiford] with his accounts for the 4th quarter 1850","year":2013,"lang":"en","type":"article","venue":"ThinkTech (Texas Tech University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Quarter (Canadian coin); Period (music); Work (physics)","score_opus":0.008198097357393535,"score_gpt":0.19695276931930603,"score_spread":0.18875467196191248,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7020581563","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013519803,0.008230899,0.00029498577,0.8537493,0.079984106,0.000017158754,0.00034947006,0.00013175367,0.055890426],"genre_scores_gemma":[0.019633994,0.0030619728,0.00023798017,0.63236254,0.03774274,0.000036342204,0.00020120598,0.00015476659,0.30656847],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9991947,0.000163963,0.00004804907,0.00016221406,0.0002466955,0.00018445143],"domain_scores_gemma":[0.9985195,0.00065959676,0.00010356484,0.00007130272,0.00046251868,0.00018351783],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010502272,0.00046401724,0.0004795785,0.0005367156,0.0067710755,0.003917686,0.0008828785,0.008985433,0.020810077],"category_scores_gemma":[0.007634497,0.00040966566,0.0004506224,0.00068871013,0.0016278768,0.0028088146,0.0011367726,0.012356811,0.011473972],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000013271857,0.0000021045157,0.00006300075,0.0000095065125,0.000001636693,0.00009113645,0.00013622758,0.000004372172,0.000022231181,0.0027436651,0.9949881,0.0019247761],"study_design_scores_gemma":[0.0000036974618,0.0000042784022,0.00044471445,0.00003624745,0.0000015788916,0.000100705605,0.00024584783,0.000020649335,0.00006108022,0.0013462715,0.9977253,0.000009574146],"about_ca_topic_score_codex":0.021769984,"about_ca_topic_score_gemma":0.044555284,"teacher_disagreement_score":0.021769984,"about_ca_system_score_codex":0.0023546661,"about_ca_system_score_gemma":0.0020288,"threshold_uncertainty_score":0.069616616},"labels":[],"label_agreement":null},{"id":"W7023356908","doi":"","title":"Ontario Cuts Solar, Wind Power Subsidies in Review - Bloomberg","year":2012,"lang":"en","type":"other","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Subsidy; Wind power; Power (physics); Government (linguistics); Boom; Renewable energy","score_opus":0.012317281740169082,"score_gpt":0.264892721266981,"score_spread":0.2525754395268119,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7023356908","genre_codex":"review","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0062411516,0.458641,0.00038544185,0.20277892,0.018889869,0.00020118555,0.03127224,0.0004742888,0.28111598],"genre_scores_gemma":[0.09119851,0.4621946,0.0013496531,0.07397466,0.0072432132,0.00029309245,0.015318637,0.00037665787,0.34805098],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99844676,0.00017540972,0.00008882552,0.000107406144,0.0008565023,0.0003251046],"domain_scores_gemma":[0.99433315,0.00083299825,0.00044617723,0.00018894766,0.003315191,0.00088362227],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022053323,0.0004915691,0.00077581673,0.0026230772,0.0012059227,0.0031173874,0.0010531151,0.001857205,0.05699372],"category_scores_gemma":[0.009943251,0.0003014567,0.0007657636,0.006618229,0.00064676,0.0013398533,0.0011774494,0.0012555904,0.0047860234],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015010433,0.000010502326,0.00091130484,0.001888816,0.00007714398,0.000035680216,0.00004461954,0.00007025306,0.000064997126,0.0030608568,0.92334646,0.07033935],"study_design_scores_gemma":[0.00007887601,0.000014853799,0.008287245,0.0021728533,0.00013956735,0.00003589048,0.000086532156,0.000032663665,0.000089970374,0.0006866058,0.98836607,0.000008806865],"about_ca_topic_score_codex":0.67339617,"about_ca_topic_score_gemma":0.87656444,"teacher_disagreement_score":0.32660383,"about_ca_system_score_codex":0.022012306,"about_ca_system_score_gemma":0.06657443,"threshold_uncertainty_score":0.65705454},"labels":[],"label_agreement":null},{"id":"W7023947339","doi":"","title":"Representing the Reprehensible: Fairy Tales, News Stories & the Monstrous Karla Homolka","year":2006,"lang":"en","type":"article","venue":"Journals @ The Mount (Mount Saint Vincent University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Meaning (existential); Personality; Public discourse; Love story","score_opus":0.012024814242046801,"score_gpt":0.22903020554153866,"score_spread":0.21700539129949187,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7023947339","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4458276,0.00687147,0.005053631,0.04584442,0.000886115,0.000033645436,0.00012630565,0.0001778471,0.49517897],"genre_scores_gemma":[0.98248416,0.00093042693,0.00034732683,0.0007789745,0.00011246877,0.000007746244,0.000020296176,0.000029736442,0.015288875],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9972601,0.001706965,0.00004851097,0.0001675226,0.00045966884,0.00035721358],"domain_scores_gemma":[0.9951474,0.0033442823,0.00046551338,0.00027317984,0.00034384136,0.0004258228],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002697854,0.00039368725,0.0002558166,0.0021758438,0.013785948,0.014887533,0.00059023517,0.0016031463,0.004238529],"category_scores_gemma":[0.0064438516,0.00026357736,0.00014909146,0.0016138854,0.02860079,0.007652498,0.0045806062,0.0024311834,0.00050173065],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000053126856,0.00003385209,0.0019848214,0.000095360505,0.000010554724,0.0005903814,0.7283748,0.00011864704,0.00060062035,0.23630217,0.010886305,0.020949392],"study_design_scores_gemma":[0.000010349635,0.000030474524,0.0035700134,0.00023503264,0.000021346796,0.00049912155,0.62804955,0.0002925623,0.0010332648,0.023873646,0.34234205,0.00004255987],"about_ca_topic_score_codex":0.02693951,"about_ca_topic_score_gemma":0.053128038,"teacher_disagreement_score":0.9730605,"about_ca_system_score_codex":0.006716277,"about_ca_system_score_gemma":0.0028787847,"threshold_uncertainty_score":0.053565383},"labels":[],"label_agreement":null},{"id":"W7023957047","doi":"","title":"A repeatable procedure to determine a representative average rail profile","year":2016,"lang":"en","type":"dissertation","venue":"Mspace (University of Manitoba)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Nucleofection; Hyporeflexia; Articular cartilage damage; Diafiltration; Dysgeusia; Fusible alloy","score_opus":0.013899201395223808,"score_gpt":0.24615642030262666,"score_spread":0.23225721890740286,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7023957047","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037025586,0.000066735796,0.9506306,0.00005513997,0.000084705614,0.0013157397,0.0009271477,0.005290415,0.004603907],"genre_scores_gemma":[0.099953905,0.00006797078,0.89366376,0.000058725378,0.000015899484,0.0010349008,0.001158483,0.0010211094,0.0030252296],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99705875,0.0004444341,0.00027758873,0.00093884626,0.0011453654,0.00013515775],"domain_scores_gemma":[0.9877467,0.0024091222,0.0008119145,0.00263966,0.006160921,0.00023161684],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004289813,0.0011980212,0.00090293784,0.0034737068,0.001507188,0.0018030436,0.0015968371,0.0010661326,0.011250367],"category_scores_gemma":[0.015625743,0.0006267731,0.00072411087,0.0021223126,0.00066142064,0.0010811671,0.0016949058,0.0015224302,0.0073964815],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004734592,0.0004827826,0.020338856,0.0008188134,0.00011852124,0.0003600538,0.0027701026,0.010796825,0.22995763,0.005653667,0.012336725,0.7158925],"study_design_scores_gemma":[0.00018175054,0.0026468125,0.16707292,0.00036222345,0.00028272232,0.0019672741,0.0034642834,0.21598482,0.47910225,0.008774284,0.11936629,0.0007944255],"about_ca_topic_score_codex":0.004105234,"about_ca_topic_score_gemma":0.010094779,"teacher_disagreement_score":0.011250367,"about_ca_system_score_codex":0.0007080147,"about_ca_system_score_gemma":0.002474788,"threshold_uncertainty_score":0.03763628},"labels":[],"label_agreement":null},{"id":"W7024176244","doi":"","title":"re:mote regina :: tom mulcaire","year":2005,"lang":"en","type":"other","venue":"Bulletin of Miscellaneous Information (Royal Gardens Kew)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Exhibition; Presentation (obstetrics); The arts; Cape; Performance art; Work (physics)","score_opus":0.0055000953988734265,"score_gpt":0.2017011791838166,"score_spread":0.1962010837849432,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7024176244","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0016068867,0.008858369,0.0006119385,0.012305033,0.0098236045,0.00014704785,0.0016078976,0.0015050103,0.96353424],"genre_scores_gemma":[0.0027199092,0.0010803384,0.00015078144,0.0005136197,0.00036045452,0.000022323606,0.00020039808,0.00034984594,0.99460226],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99972755,0.000028760642,0.000006876456,0.00006645561,0.00009933948,0.00007091772],"domain_scores_gemma":[0.9995763,0.000035248737,0.000030383064,0.00003690117,0.0001388592,0.00018239961],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0004151808,0.0012603344,0.00043812057,0.0007199466,0.002157843,0.003381512,0.00074805826,0.0014791841,0.7404896],"category_scores_gemma":[0.0012895511,0.00040209596,0.00030534194,0.00056304224,0.00040189555,0.002516372,0.0028936672,0.0015605161,0.43520302],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000019065925,0.000009936285,0.00006292617,0.00006785473,0.0000011120286,0.00008936714,0.00012896745,0.000014317081,0.0003646798,0.00071131054,0.9738709,0.0246596],"study_design_scores_gemma":[0.000001612594,0.0000035878131,0.000134223,0.00002813067,4.986326e-7,0.00005232172,0.00009464044,0.000006542283,0.000070986454,0.00005189168,0.99955386,0.0000018074893],"about_ca_topic_score_codex":0.0076702978,"about_ca_topic_score_gemma":0.021761296,"teacher_disagreement_score":0.2595104,"about_ca_system_score_codex":0.0010220425,"about_ca_system_score_gemma":0.0008763173,"threshold_uncertainty_score":0.37016004},"labels":[],"label_agreement":null},{"id":"W7024237576","doi":"","title":"The road between now and then","year":2001,"lang":"en","type":"other","venue":"Library and Archives Canada (Government of Canada)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Government (linguistics); Work (physics); Natural (archaeology); Perspective (graphical)","score_opus":0.003395538986901898,"score_gpt":0.16150011742275414,"score_spread":0.15810457843585224,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7024237576","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0030531634,0.0107910205,0.0013794338,0.15550224,0.0042460063,0.000020501606,0.00038523728,0.00019341172,0.824429],"genre_scores_gemma":[0.09496354,0.007013082,0.001601629,0.02383415,0.0005074475,0.00003592559,0.00035397755,0.00049169903,0.87119853],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9964059,0.00039597426,0.00006945519,0.0004245721,0.001296616,0.0014074016],"domain_scores_gemma":[0.99678075,0.00027724155,0.00006973126,0.00024391881,0.0012603913,0.0013678878],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023160856,0.0005853639,0.0007195395,0.0017883759,0.025959047,0.024829429,0.0017062317,0.005481169,0.122983746],"category_scores_gemma":[0.0051216925,0.0005222248,0.0004465722,0.0028263435,0.013118067,0.018390786,0.007198639,0.012561836,0.02170123],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003255874,0.000021116542,0.00060267776,0.000061642495,0.0000036780818,0.00006831719,0.0054695755,0.00005649194,0.00012259647,0.50985533,0.41159406,0.07211205],"study_design_scores_gemma":[0.000001986612,0.0000032177084,0.0005300777,0.00013480206,0.0000020231453,0.000036813904,0.004524679,0.000020696807,0.000044960867,0.017500587,0.97718906,0.000011138087],"about_ca_topic_score_codex":0.8230835,"about_ca_topic_score_gemma":0.90978676,"teacher_disagreement_score":0.8230835,"about_ca_system_score_codex":0.044867724,"about_ca_system_score_gemma":0.06894206,"threshold_uncertainty_score":0.41142166},"labels":[],"label_agreement":null},{"id":"W7024299326","doi":"","title":"The role of agricultural policies in rural development/ comparative study between Iran and Japan","year":2018,"lang":"en","type":"article","venue":"DOAJ (DOAJ: Directory of Open Access Journals)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Agriculture; Quarter (Canadian coin); Agricultural policy; Agricultural productivity; Rural area; Theme (computing); Rural economy; Rural development","score_opus":0.1967255651573538,"score_gpt":0.5365624480065508,"score_spread":0.33983688284919705,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7024299326","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9946131,0.00049898966,0.000024497944,0.00019984737,0.0000034782506,0.000004224253,0.00002762757,8.233003e-7,0.004627332],"genre_scores_gemma":[0.9989115,0.00053128734,0.000038090388,0.000036294525,0.000004950011,0.000003400406,0.000027607783,8.3398106e-7,0.0004459756],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9997004,0.00008447844,0.000014644592,0.000023810086,0.000041391777,0.00013535467],"domain_scores_gemma":[0.99908245,0.00016314494,0.0003282823,0.000024638624,0.00014721468,0.00025428392],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000607759,0.00010532908,0.00016007014,0.0011106138,0.0008443185,0.0010426905,0.00017714906,0.00015242407,0.0014991823],"category_scores_gemma":[0.00089577184,0.000088324814,0.00016365746,0.002544218,0.0009148346,0.0005928924,0.0007182964,0.00023907976,0.000078768],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003085714,0.000447681,0.9117955,0.0003654803,0.000109877205,0.0017894007,0.03313501,0.0006308541,0.0017661426,0.011003167,0.0012005153,0.037447818],"study_design_scores_gemma":[0.00000933539,0.000105549276,0.9385923,0.000046624544,0.000034611294,0.00017380937,0.055230215,0.00017136981,0.00015174865,0.00035502555,0.0051197964,0.000009634042],"about_ca_topic_score_codex":0.028698988,"about_ca_topic_score_gemma":0.051558822,"teacher_disagreement_score":0.028698988,"about_ca_system_score_codex":0.0016958896,"about_ca_system_score_gemma":0.0023529227,"threshold_uncertainty_score":0.057063878},"labels":[],"label_agreement":null},{"id":"W7024327003","doi":"","title":"Quest : finding form in children's literature","year":2012,"lang":"en","type":"other","venue":"OpenGrey (Institut de l'Information Scientifique et Technique)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Perspective (graphical); Element (criminal law); Character (mathematics); Fantasy; Context (archaeology); Order (exchange)","score_opus":0.01194257135248826,"score_gpt":0.2796330711531889,"score_spread":0.26769049980070064,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7024327003","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20409584,0.015416463,0.011998958,0.008864711,0.00030402243,0.00009895376,0.0003451731,0.00016962484,0.75870633],"genre_scores_gemma":[0.9676301,0.004296226,0.0039343387,0.0003161606,0.000121306424,0.00006945265,0.00016343719,0.00009872163,0.023370164],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985505,0.00076162355,0.000074695505,0.0001690311,0.00027503993,0.00016898017],"domain_scores_gemma":[0.99745387,0.0015009046,0.00034924824,0.00026177798,0.00021282826,0.00022140496],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015647628,0.00033000388,0.00030725115,0.0032167595,0.005341224,0.011810238,0.0007323299,0.0011699053,0.005764899],"category_scores_gemma":[0.004468817,0.00029187664,0.00027552646,0.0034367756,0.020060219,0.0086050825,0.004466584,0.0015359545,0.00061914755],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000018034332,0.00001339895,0.0016729084,0.0001277059,0.0000042203383,0.0005037736,0.17271106,0.00009089171,0.00020007919,0.80441797,0.003942798,0.016297162],"study_design_scores_gemma":[0.000015239257,0.000025539992,0.0037029772,0.0008019469,0.000010983001,0.001974551,0.24622867,0.0002949223,0.00060672197,0.20774439,0.5385687,0.00002529266],"about_ca_topic_score_codex":0.0054569216,"about_ca_topic_score_gemma":0.0109101515,"teacher_disagreement_score":0.011810238,"about_ca_system_score_codex":0.0049253493,"about_ca_system_score_gemma":0.002421469,"threshold_uncertainty_score":0.035736144},"labels":[],"label_agreement":null},{"id":"W7024380295","doi":"","title":"8,789 Shares in Canadian Solar Inc. (NASDAQ:CSIQ) Bought by XTX Topco Ltd","year":2021,"lang":"en","type":"other","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Government (linguistics); Consumption (sociology)","score_opus":0.006976955539238327,"score_gpt":0.25498128354957533,"score_spread":0.248004328010337,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7024380295","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0026136823,0.0009158147,0.00062354107,0.0015006224,0.00084853516,0.00010136134,0.015364695,0.001490449,0.9765413],"genre_scores_gemma":[0.0024322078,0.00022311968,0.00012388814,0.00012459075,0.000049385482,0.000008699565,0.0028453867,0.00019191904,0.99400085],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99946696,0.000016004755,0.000011169263,0.000080096244,0.00029475108,0.00013096748],"domain_scores_gemma":[0.99870455,0.00002883071,0.000030223631,0.00008665011,0.0006742891,0.00047536998],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.00049119646,0.0011570895,0.0005903741,0.0020860774,0.0029293797,0.003932052,0.0009817292,0.0009447096,0.60387444],"category_scores_gemma":[0.0010390587,0.00044115726,0.00055095623,0.002908571,0.00080988824,0.0012084962,0.0016053658,0.0012436054,0.46970505],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000083563,0.00004330497,0.00043883015,0.00005567871,0.000004367528,0.000040897576,0.000033012584,0.00012031936,0.0008000594,0.0033027516,0.9407265,0.054350793],"study_design_scores_gemma":[0.000010775589,0.000013198499,0.0012898579,0.000019974634,0.0000028387417,0.00002247366,0.00006030816,0.00014631764,0.00032850372,0.00027011492,0.99783033,0.0000053241843],"about_ca_topic_score_codex":0.33397612,"about_ca_topic_score_gemma":0.53705794,"teacher_disagreement_score":0.66602385,"about_ca_system_score_codex":0.0067229033,"about_ca_system_score_gemma":0.011988325,"threshold_uncertainty_score":0.6640643},"labels":[],"label_agreement":null},{"id":"W7024396789","doi":"","title":"Role of miRNAs in Translational Control of Human Apolipoprotein B-100 mRNA","year":2013,"lang":"en","type":"dissertation","venue":"Library and Archives Canada (Government of Canada)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Apolipoprotein B; Messenger RNA; Transfection; microRNA; Translational efficiency; Translation (biology); Three prime untranslated region; Translational regulation","score_opus":0.002654248860639226,"score_gpt":0.172964819059867,"score_spread":0.17031057019922777,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7024396789","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98028207,0.0072320797,0.0048491713,0.0002184508,0.000061562365,0.000036061865,0.00023423569,0.000089914116,0.006996461],"genre_scores_gemma":[0.98603594,0.0029668882,0.0035931477,0.00008147236,0.000017103255,0.000030218953,0.00027477552,0.000025177922,0.0069752308],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99988806,0.000017168986,0.000008347538,0.00003809397,0.000030326502,0.000018049865],"domain_scores_gemma":[0.9999434,0.000013091131,0.000012907742,0.000006233936,0.0000148399395,0.000009687877],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00015482875,0.00016188982,0.00013831421,0.00013633046,0.0002093894,0.00040624687,0.0000792991,0.00014073902,0.0007711374],"category_scores_gemma":[0.00018037611,0.000094626244,0.00019045963,0.00006652777,0.00018034568,0.00010592102,0.00015402271,0.0002539728,0.00040812325],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018702885,0.000021094733,0.00086912455,0.00003570159,0.000004766962,0.00011414357,0.00007051173,0.00009375567,0.991676,0.0003400575,0.00008173757,0.0065060905],"study_design_scores_gemma":[0.000019154606,0.00019202792,0.01906237,0.000013709037,0.000025132193,0.00030336488,0.00010699151,0.0011249968,0.97099775,0.00023161751,0.007914451,0.000008502804],"about_ca_topic_score_codex":0.0013071016,"about_ca_topic_score_gemma":0.0014017292,"teacher_disagreement_score":0.0013071016,"about_ca_system_score_codex":0.00036581975,"about_ca_system_score_gemma":0.00040223994,"threshold_uncertainty_score":0.0026542544},"labels":[],"label_agreement":null},{"id":"W7024428693","doi":"","title":"Seattle Mariners vs Toronto Blue Jays Live Broadcast Free Online","year":2022,"lang":"en","type":"other","venue":"OSF Preprints (OSF Preprints)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Broadcasting (networking); The Internet; Doors; Public access","score_opus":0.010645282151407951,"score_gpt":0.2635796097322182,"score_spread":0.25293432758081025,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7024428693","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039420493,0.00037329795,0.00042430215,0.004147917,0.0015136731,0.00005797142,0.020504631,0.0014039325,0.9321537],"genre_scores_gemma":[0.082915865,0.00022669474,0.0004132702,0.0005272952,0.00021619815,0.00002714981,0.006871035,0.00066505116,0.90813744],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99969244,0.000023915352,0.0000047651542,0.000058373844,0.00012449692,0.00009601824],"domain_scores_gemma":[0.9985098,0.00024304928,0.00009057229,0.00015199339,0.0003684544,0.0006360738],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0003698078,0.00039293096,0.00025451503,0.0010693596,0.0026619264,0.0025293794,0.0006436682,0.0010614834,0.51294994],"category_scores_gemma":[0.0018182842,0.00023879818,0.0002199991,0.001077846,0.0005177478,0.0011065206,0.0014238355,0.0012149926,0.09392588],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00061311206,0.00009079833,0.0028918942,0.0000924278,0.000011875941,0.00018766543,0.0005869281,0.00017086805,0.0009451513,0.003442975,0.9429215,0.048044764],"study_design_scores_gemma":[0.00009046163,0.00008693057,0.029178275,0.000114486684,0.000017410306,0.00009476302,0.0043376633,0.00036938887,0.0015002827,0.00068105204,0.9635067,0.000022574379],"about_ca_topic_score_codex":0.20262314,"about_ca_topic_score_gemma":0.46178737,"teacher_disagreement_score":0.48705006,"about_ca_system_score_codex":0.002222713,"about_ca_system_score_gemma":0.0022680946,"threshold_uncertainty_score":0.69471776},"labels":[],"label_agreement":null},{"id":"W7024522294","doi":"","title":"Share original pro max real","year":2022,"lang":"ar","type":"other","venue":"Bulletin of Miscellaneous Information (Royal Gardens Kew)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Brother; Supporter; Spanish Civil War; Hinduism; Clan","score_opus":0.008750522768568851,"score_gpt":0.21705696735389046,"score_spread":0.2083064445853216,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7024522294","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00031339406,0.0005195121,0.0063855853,0.0027023253,0.004163744,0.00043536042,0.047167677,0.08176234,0.85655004],"genre_scores_gemma":[0.002312199,0.00059956516,0.0032863726,0.0014207809,0.0007974911,0.0005844195,0.040612318,0.055146746,0.89524007],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9981127,0.00022694554,0.00011779047,0.0004238751,0.0007565313,0.00036224662],"domain_scores_gemma":[0.9943474,0.00065717415,0.00016729397,0.0018102323,0.0016123572,0.0014056277],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.001661155,0.002431985,0.0020722332,0.004445682,0.0024061736,0.015162722,0.0055435966,0.0038324909,0.9543723],"category_scores_gemma":[0.013128819,0.0014487519,0.0024363664,0.005464556,0.0013373094,0.00933336,0.010523568,0.0029379032,0.9626174],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000051510768,0.000013815376,0.00003262056,0.00017661856,0.000004056673,0.00003073209,0.00004326189,0.0000598969,0.00021565489,0.0012553368,0.9776053,0.020511273],"study_design_scores_gemma":[0.000021357848,0.000007928334,0.00010541735,0.00005471589,0.0000028873737,0.000040475763,0.000055426262,0.00009453589,0.00021300721,0.0011579294,0.9982297,0.000016642536],"about_ca_topic_score_codex":0.0019035556,"about_ca_topic_score_gemma":0.002566149,"teacher_disagreement_score":0.045627713,"about_ca_system_score_codex":0.0017498628,"about_ca_system_score_gemma":0.0025311643,"threshold_uncertainty_score":0.06508231},"labels":[],"label_agreement":null},{"id":"W7024575673","doi":"","title":"Séquence des cours de français et approche-programme","year":2015,"lang":"fr","type":"other","venue":"Bibliothèque et Archives nationales du Québec (Québec government)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Collège de Bois-de-Boulogne","funders":"","keywords":"Nucleofection; Gestational period; TSG101; Dysgeusia; Liquation; Diafiltration; Emperipolesis; Triacetin; Fusible alloy","score_opus":0.020362948217831175,"score_gpt":0.2574546841429343,"score_spread":0.23709173592510313,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7024575673","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12493139,0.010205889,0.049433973,0.013884935,0.006198293,0.0018414336,0.080572896,0.010972287,0.7019589],"genre_scores_gemma":[0.21000598,0.004784966,0.048941657,0.0018287394,0.0006679041,0.0010411891,0.06735862,0.0035557572,0.6618151],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9949268,0.000984627,0.00022811221,0.0006739173,0.002518772,0.0006678109],"domain_scores_gemma":[0.98887,0.0015097905,0.00037561858,0.0009851335,0.0075655617,0.0006939177],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004213649,0.0011512345,0.0004929094,0.0057654157,0.0032436147,0.0053403093,0.0007916771,0.0013400549,0.0445055],"category_scores_gemma":[0.012816181,0.00046910907,0.000659625,0.006043218,0.00078615424,0.0014246703,0.0012224994,0.001661272,0.01666222],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00091945264,0.00024852817,0.011077953,0.00088595843,0.00010587614,0.00060203794,0.0023652026,0.0043702805,0.0066314456,0.02799788,0.6081685,0.33662677],"study_design_scores_gemma":[0.00006319055,0.00006700336,0.014436996,0.00021922508,0.00002510212,0.000116358846,0.0010546186,0.0020360593,0.0027662204,0.001098039,0.97806895,0.00004821618],"about_ca_topic_score_codex":0.64548665,"about_ca_topic_score_gemma":0.6322871,"teacher_disagreement_score":0.64548665,"about_ca_system_score_codex":0.012374201,"about_ca_system_score_gemma":0.0145663265,"threshold_uncertainty_score":0.71320224},"labels":[],"label_agreement":null},{"id":"W7024576020","doi":"","title":"Shpenopholis intermedia (Slender Wedge grass)","year":2004,"lang":"en","type":"other","venue":"Connecting Canadians: Canada’s Multicultural Newspapers Beta Website (Athabasca University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Wedge (geometry); Focus (optics); Context (archaeology); Deformation (meteorology)","score_opus":0.0071447203416480295,"score_gpt":0.19529153593622825,"score_spread":0.1881468155945802,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7024576020","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3195287,0.011831141,0.0040030624,0.0026288526,0.0005284997,0.0005128578,0.006533412,0.0011990019,0.6532344],"genre_scores_gemma":[0.6469177,0.0060736216,0.005077932,0.0012761788,0.00008119087,0.000074594725,0.0020918085,0.00013006499,0.33827695],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9998841,0.0000047385447,0.0000025570212,0.000027337468,0.000048086124,0.00003322456],"domain_scores_gemma":[0.9998889,0.000008036879,0.000010697485,0.0000048315705,0.000048073423,0.000039439903],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00011301141,0.00060654146,0.00034071345,0.001882899,0.0026739952,0.00056887727,0.00047175906,0.00046431992,0.02987616],"category_scores_gemma":[0.00013149905,0.00014755286,0.00022000413,0.0009665847,0.0005220566,0.0003549422,0.0005689882,0.00062075444,0.005329298],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015713935,0.00059997954,0.0080344975,0.0015254067,0.00006815418,0.0035947387,0.0015540558,0.0011898349,0.38026077,0.009049979,0.08124535,0.51130587],"study_design_scores_gemma":[0.00017671564,0.00088152423,0.08585351,0.00042834066,0.00018234328,0.0019365164,0.0026827494,0.0010419524,0.06872576,0.0021137702,0.83588815,0.000088760054],"about_ca_topic_score_codex":0.5104406,"about_ca_topic_score_gemma":0.7558185,"teacher_disagreement_score":0.5104406,"about_ca_system_score_codex":0.0022750879,"about_ca_system_score_gemma":0.003112288,"threshold_uncertainty_score":0.98488504},"labels":[],"label_agreement":null},{"id":"W7024603552","doi":"","title":"Sound transmission loss of orthotropic sandwich panels with soft core and noise control treatment","year":2012,"lang":"en","type":"article","venue":"NPARC","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Shearing (physics); Sandwich-structured composite; Noise control; Sound transmission class; Orthotropic material; Compressibility; Core (optical fiber); Noise (video)","score_opus":0.017896965668736824,"score_gpt":0.26591228234161557,"score_spread":0.24801531667287874,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7024603552","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37392905,0.00057112967,0.6119567,0.00013236012,0.00007960651,0.000035745117,0.00010264463,0.00025632165,0.01293633],"genre_scores_gemma":[0.98521674,0.00019906553,0.010003236,0.000026895204,0.000010882449,0.000024462732,0.000037309965,0.000022945876,0.004458363],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99977654,0.000060329057,0.0000063579687,0.00003431642,0.00010339108,0.00001908485],"domain_scores_gemma":[0.99982005,0.00006940148,0.000029029121,0.000033047905,0.000035150973,0.000013265199],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00036766665,0.0005382062,0.00045200507,0.0002598022,0.00019315016,0.0005652072,0.0006010426,0.00088607374,0.00095683074],"category_scores_gemma":[0.0005787194,0.00021712128,0.00051480054,0.00018673162,0.00056450657,0.0006745013,0.00038933582,0.00043924397,0.0002488685],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015404371,0.0000672332,0.0008718269,0.000085639505,0.000024517187,0.000489935,0.000090098474,0.89061046,0.08842855,0.0085235145,0.00034527242,0.010308998],"study_design_scores_gemma":[0.000002749554,0.00002497649,0.00030639695,0.0000030854014,0.0000061390438,0.000050957042,0.000009277011,0.9945045,0.0044492385,0.0005344058,0.00010288149,0.0000053781214],"about_ca_topic_score_codex":0.0009941667,"about_ca_topic_score_gemma":0.00086393254,"teacher_disagreement_score":0.0009941667,"about_ca_system_score_codex":0.00033224205,"about_ca_system_score_gemma":0.00024460148,"threshold_uncertainty_score":0.0032009482},"labels":[],"label_agreement":null},{"id":"W7024798424","doi":"","title":"Smoke from Canada wildfires increasing health risks in Black and poorer US communities","year":2023,"lang":"en","type":"other","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Smoke; Air pollution; Human health; Public health; Poison control","score_opus":0.03799644492927861,"score_gpt":0.3071918132148642,"score_spread":0.2691953682855856,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7024798424","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9932914,0.00037702537,0.000048321603,0.00074624014,0.000022295799,0.000018596302,0.0017646834,0.0000057694087,0.0037257338],"genre_scores_gemma":[0.99677914,0.0003616281,0.000051760908,0.00020602296,0.000018409231,0.000010831139,0.0005458217,0.0000060668017,0.0020203386],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9995881,0.000046086643,0.000017200178,0.000057987036,0.00008345261,0.00020730952],"domain_scores_gemma":[0.9979572,0.00013096747,0.0004957323,0.000059338436,0.0004432261,0.000913417],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00025378083,0.00027913638,0.00025306756,0.0011059495,0.0022244852,0.0010576985,0.00069256744,0.00083117525,0.0067755277],"category_scores_gemma":[0.0022399633,0.00026937563,0.00060777494,0.0015205374,0.000584984,0.0005634852,0.0012833758,0.0012228354,0.0003431598],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013782976,0.00008262065,0.99346113,0.00001859217,0.00004482049,0.000060124024,0.00076879235,0.00006081656,0.000111513255,0.00006380248,0.001311468,0.0038783355],"study_design_scores_gemma":[0.0000054784177,0.000024860845,0.99428517,0.000038632974,0.000030071242,0.000030144116,0.004765957,0.00014693318,0.00002966421,0.00006084199,0.00057371316,0.00000852337],"about_ca_topic_score_codex":0.9390347,"about_ca_topic_score_gemma":0.96929604,"teacher_disagreement_score":0.0609653,"about_ca_system_score_codex":0.003688058,"about_ca_system_score_gemma":0.005271861,"threshold_uncertainty_score":0.1226486},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"observational","genre":"empirical","about_ca_system":false,"about_ca_topic":true,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"observational","genre":"empirical","about_ca_system":false,"about_ca_topic":true,"confidence":"low"}],"label_agreement":"agree"},{"id":"W7024804035","doi":"","title":"A supported college course for credit: More than just Academics","year":2017,"lang":"en","type":"other","venue":"Arca (British Columbia Electronic Library Network)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Mental health; Session (web analytics); Presentation (obstetrics); Coaching; Academic advising; Higher education","score_opus":0.011443729804320837,"score_gpt":0.2542931138448453,"score_spread":0.24284938404052447,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7024804035","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006965379,0.0014322976,0.001528109,0.03612137,0.031034658,0.00089060824,0.0040556174,0.0026947556,0.9152773],"genre_scores_gemma":[0.009487892,0.0007178388,0.0007944133,0.0039106295,0.0018822306,0.00019161482,0.001818125,0.0005473518,0.9806498],"study_design_codex":"not_applicable","study_design_gemma":"qualitative","domain_scores_codex":[0.99884975,0.00011477713,0.00004866537,0.00016426975,0.0005108964,0.00031151198],"domain_scores_gemma":[0.9914396,0.000112808164,0.0001145299,0.00035179226,0.0019136767,0.0060676117],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0012539279,0.0014981827,0.0010545026,0.00074469746,0.004470094,0.009631332,0.0020737406,0.0018788456,0.63452226],"category_scores_gemma":[0.0034459517,0.00047436275,0.00063670106,0.00074103434,0.0008483895,0.0039073075,0.008109621,0.0040747058,0.39245263],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000034460343,0.0001454753,0.00015356478,0.000043662858,0.0000013037472,0.000037582347,0.00008454634,0.00001889543,0.00025062272,0.0005901186,0.9713192,0.027320597],"study_design_scores_gemma":[0.00001523553,0.0000623731,0.0015229989,0.00008129077,0.000002279473,0.000044734978,0.00044153104,0.000043703818,0.00008756011,0.00057151605,0.99711597,0.00001077487],"about_ca_topic_score_codex":0.008336197,"about_ca_topic_score_gemma":0.04349701,"teacher_disagreement_score":0.63452226,"about_ca_system_score_codex":0.0030345528,"about_ca_system_score_gemma":0.005757688,"threshold_uncertainty_score":0.5213096},"labels":[],"label_agreement":null},{"id":"W7025005265","doi":"","title":"Ãtude de la conversion de longueur d'onde","year":2000,"lang":"fr","type":"other","venue":"Library and Archives Canada (Government of Canada)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Nucleofection; Gestational period; TSG101; Dysgeusia; Diafiltration; Liquation; Emperipolesis; Triacetin; Fusible alloy","score_opus":0.002459851744428148,"score_gpt":0.15972918954532722,"score_spread":0.15726933780089908,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7025005265","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32114148,0.011837098,0.049543872,0.010779799,0.0014566053,0.0006685425,0.00394197,0.004672922,0.59595764],"genre_scores_gemma":[0.5929614,0.006566131,0.024841923,0.0014878454,0.00027445197,0.0002881429,0.0030948073,0.001222275,0.36926305],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9933687,0.0013752502,0.00024352237,0.0006883366,0.003658761,0.0006655027],"domain_scores_gemma":[0.991907,0.002416411,0.000317379,0.0010493698,0.0040026703,0.00030713118],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004400911,0.0011233274,0.0007918816,0.0040960177,0.007223395,0.011719456,0.001937742,0.0025903473,0.03359999],"category_scores_gemma":[0.020758903,0.000819802,0.0013594332,0.003935712,0.0034720828,0.0035729513,0.00193153,0.004710994,0.0065612523],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008102483,0.000804916,0.026165066,0.00064753694,0.00013559088,0.0022220144,0.011338316,0.0069183926,0.0093272,0.14389499,0.044186432,0.75354934],"study_design_scores_gemma":[0.00016664139,0.00022358086,0.08098961,0.0009197261,0.00013051722,0.004314894,0.012801158,0.013728589,0.027304368,0.007541887,0.8516667,0.00021235643],"about_ca_topic_score_codex":0.8482302,"about_ca_topic_score_gemma":0.7589458,"teacher_disagreement_score":0.8482302,"about_ca_system_score_codex":0.02788996,"about_ca_system_score_gemma":0.024165496,"threshold_uncertainty_score":0.30532718},"labels":[],"label_agreement":null},{"id":"W7025322670","doi":"","title":"Uplatnění amerických plemen koní v České republice","year":2011,"lang":"en","type":"dissertation","venue":"Digital Repository (National Repository of Grey Literature)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Czech; Quarter (Canadian coin); Work (physics); International comparisons; Western europe","score_opus":0.009545500508091054,"score_gpt":0.2555883167203818,"score_spread":0.24604281621229074,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7025322670","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10522884,0.08509956,0.00843232,0.007306402,0.0046383003,0.00027883664,0.011248451,0.0010091381,0.77675813],"genre_scores_gemma":[0.39847174,0.05558879,0.010705531,0.0010281606,0.0007724218,0.00033678042,0.009605779,0.00090571895,0.52258503],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.999561,0.00004244444,0.000033734294,0.00011987994,0.00015601094,0.00008699454],"domain_scores_gemma":[0.9997212,0.00003427377,0.000033505243,0.000031915606,0.00010538354,0.00007373224],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00047483662,0.00048776108,0.0006673014,0.0017815789,0.0020402807,0.0039850664,0.0004886569,0.0005563696,0.043100238],"category_scores_gemma":[0.0010097917,0.0002569202,0.0003991939,0.0021012682,0.00091620587,0.0012999399,0.0019436576,0.0017862142,0.01640554],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043268062,0.00022288445,0.010726595,0.0033978256,0.00008629185,0.0011740427,0.007519966,0.0022686338,0.010540415,0.2062275,0.23561332,0.52178985],"study_design_scores_gemma":[0.000008785408,0.00003969481,0.012094648,0.0005399426,0.000015497137,0.00035792647,0.000914247,0.00011481638,0.001152812,0.0028327077,0.981911,0.000017921977],"about_ca_topic_score_codex":0.0053366763,"about_ca_topic_score_gemma":0.012404312,"teacher_disagreement_score":0.043100238,"about_ca_system_score_codex":0.0018246346,"about_ca_system_score_gemma":0.004266476,"threshold_uncertainty_score":0.14418465},"labels":[],"label_agreement":null},{"id":"W7035770051","doi":"","title":"Ajustement d’enseignantes et d’enseignants immigrants aux conventions professionnelles de l’École québécoise au cœur de l’accompagnement offert par des conseillères pédagogiques","year":2022,"lang":"fr","type":"dissertation","venue":"Papyrus : Institutional Repository (Université de Montréal)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Immigration; Context (archaeology); Citizenship; Naturalization","score_opus":0.020241809669695736,"score_gpt":0.24716665456113623,"score_spread":0.2269248448914405,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7035770051","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9510973,0.0013003509,0.0008082683,0.009016082,0.00025264625,0.000068757865,0.00016300777,0.00003112595,0.037262447],"genre_scores_gemma":[0.9576512,0.00090530113,0.00089393486,0.0019800288,0.000031629534,0.000069779046,0.00015641433,0.000028251849,0.0382834],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.997419,0.00077433017,0.00006537927,0.00029840934,0.000637859,0.0008049027],"domain_scores_gemma":[0.99327844,0.0006984398,0.0007678329,0.00028319648,0.001961815,0.0030103072],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003068747,0.0003364809,0.00038526137,0.00091072574,0.01190251,0.0053866263,0.0011699481,0.0014517689,0.012454726],"category_scores_gemma":[0.007208194,0.00026365175,0.00028249182,0.0010171647,0.0035505937,0.0024089818,0.003929809,0.0029384003,0.0012956194],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025440712,0.0003500848,0.22066908,0.00038469542,0.00006720198,0.0012295551,0.5922242,0.0001955858,0.0035753546,0.020506267,0.027228314,0.13331525],"study_design_scores_gemma":[0.000029240946,0.00015646216,0.32164976,0.00062234537,0.00005503454,0.00023237469,0.4989713,0.0004608214,0.0005735447,0.0021588043,0.17499915,0.00009114059],"about_ca_topic_score_codex":0.72129077,"about_ca_topic_score_gemma":0.88911253,"teacher_disagreement_score":0.98302174,"about_ca_system_score_codex":0.016978288,"about_ca_system_score_gemma":0.03165843,"threshold_uncertainty_score":0.5607012},"labels":[],"label_agreement":null},{"id":"W7084159128","doi":"10.5194/egusphere-2025-1572-ac3","title":"Reply on RC3","year":2025,"lang":"en","type":"peer-review","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université du Québec à Rimouski; Center for Northern Studies; Nordic Life Science Pipeline (Canada); Université Laval","funders":"","keywords":"Context (archaeology); Snow; Tropical cyclone forecast model; Weather forecasting; Anticipation (artificial intelligence); Hazard; Global Forecast System; Christian ministry","score_opus":0.02589560190014495,"score_gpt":0.34924923993360124,"score_spread":0.3233536380334563,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7084159128","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0003374987,0.0013652686,0.00037653052,0.7816473,0.18141107,0.00017496574,0.0012574092,0.00065350527,0.032776356],"genre_scores_gemma":[0.004302152,0.0012438081,0.00032307894,0.7409577,0.059204485,0.00024099032,0.00055429485,0.0003962257,0.1927772],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9978492,0.00043250067,0.00025086306,0.00033194866,0.0007875341,0.0003479134],"domain_scores_gemma":[0.990193,0.0027590755,0.0003559865,0.0004778076,0.0047857896,0.001428339],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023472703,0.00067096413,0.0008613889,0.0010829128,0.0024312134,0.0031880098,0.002338404,0.016887318,0.19602983],"category_scores_gemma":[0.035321843,0.0003555552,0.0010607146,0.0007701482,0.0014068861,0.0026920573,0.0019892645,0.012694158,0.13687769],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000011355809,0.0000029377254,0.00004059236,0.000013668633,0.0000012652979,0.000052734653,0.000009199521,0.0000035772973,0.000016626263,0.00015959545,0.9978434,0.0018450517],"study_design_scores_gemma":[0.000010660875,0.00000884267,0.00020408124,0.000057204372,0.0000029653518,0.000090161855,0.000070340444,0.000029506558,0.00007318616,0.0003294221,0.9991142,0.000009362362],"about_ca_topic_score_codex":0.009902064,"about_ca_topic_score_gemma":0.01190135,"teacher_disagreement_score":0.19602983,"about_ca_system_score_codex":0.0032343315,"about_ca_system_score_gemma":0.0031414127,"threshold_uncertainty_score":0.6557851},"labels":[],"label_agreement":null},{"id":"W7085121138","doi":"10.1109/raiic65850.2025.11170191","title":"Application of Vision-Language Models to Pedestrian Behavior Prediction and Scene Understanding in Autonomous Driving","year":2025,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Pedestrian; Metric (unit); Perception; Trajectory; Key (lock); Advanced driver assistance systems","score_opus":0.01962526538052656,"score_gpt":0.30733318071728566,"score_spread":0.2877079153367591,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7085121138","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.35415906,0.0008817674,0.63485754,0.0010308311,0.00015395619,0.00008665167,0.00045949593,0.0052158018,0.0031548906],"genre_scores_gemma":[0.9421063,0.00014268121,0.055710252,0.00016933879,0.000027162272,0.000038923525,0.00047552175,0.00007865468,0.0012510958],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997204,0.00007430041,0.000010609634,0.00010507286,0.000043142176,0.000046431665],"domain_scores_gemma":[0.99927896,0.00038447173,0.000053157153,0.000087134045,0.00014964715,0.000046735662],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007085786,0.0009654629,0.0005966247,0.0006091975,0.00042869896,0.00074127276,0.0010887443,0.000940882,0.00092858216],"category_scores_gemma":[0.0026039283,0.00043609444,0.000744002,0.0004098442,0.00041845907,0.0017543221,0.00090925395,0.0018485571,0.00034079433],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022492613,0.0003503004,0.0026766679,0.00006227375,0.00013229549,0.00011896477,0.0001803252,0.78564364,0.010400332,0.0025188434,0.0022789384,0.19541244],"study_design_scores_gemma":[0.0000031014013,0.000019176578,0.00018828594,0.0000022252145,0.000006694754,0.0000064694564,0.000014456271,0.9965083,0.001563683,0.0015507363,0.00013223158,0.0000046394084],"about_ca_topic_score_codex":0.023110775,"about_ca_topic_score_gemma":0.024384655,"teacher_disagreement_score":0.023110775,"about_ca_system_score_codex":0.00094304717,"about_ca_system_score_gemma":0.0011055729,"threshold_uncertainty_score":0.0459525},"labels":[],"label_agreement":null},{"id":"W7087312099","doi":"10.1007/978-3-032-07845-2_18","title":"Test-Time Adaptation of Medical Vision-Language Models","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Adaptation (eye); Benchmark (surveying); Domain adaptation; Downstream (manufacturing); Mutual information; Code (set theory); Class (philosophy); Transfer of learning","score_opus":0.011620095933489439,"score_gpt":0.2857211316042886,"score_spread":0.27410103567079913,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7087312099","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21593983,0.0032661327,0.75410944,0.00070933247,0.0006192462,0.00021767126,0.0020467609,0.014152447,0.008939088],"genre_scores_gemma":[0.8687127,0.00062648434,0.11107719,0.0005335509,0.00015923595,0.00015697908,0.0049022613,0.0010718778,0.012759619],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992353,0.00026864966,0.00003625923,0.00021243245,0.00015085307,0.00009664233],"domain_scores_gemma":[0.9981957,0.0010661029,0.00006391907,0.00030083992,0.00029942795,0.00007411913],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016875812,0.0008579297,0.0006419582,0.00050806714,0.00016316712,0.0007845411,0.0012442316,0.0012593961,0.0040987944],"category_scores_gemma":[0.00611656,0.00036246274,0.00069248595,0.0005647821,0.0002725202,0.0008006781,0.0010347451,0.0011033204,0.003112717],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012535928,0.0003325573,0.0028552967,0.00014991248,0.00025686398,0.0002416429,0.00006451038,0.17823401,0.049815297,0.00085975626,0.009233705,0.7567029],"study_design_scores_gemma":[0.000023681283,0.000100690726,0.0027837523,0.000009450537,0.000057072684,0.00019504054,0.000017888497,0.9761329,0.01849138,0.00081651396,0.0013515676,0.000020120098],"about_ca_topic_score_codex":0.006047654,"about_ca_topic_score_gemma":0.006076661,"teacher_disagreement_score":0.006047654,"about_ca_system_score_codex":0.00048263255,"about_ca_system_score_gemma":0.0005166474,"threshold_uncertainty_score":0.01371187},"labels":[],"label_agreement":null},{"id":"W7093322906","doi":"10.1145/3746276.3760469","title":"TemporalCook: Benchmarking Temporal and Procedural Reasoning in Multimodal Large Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Benchmark (surveying); Benchmarking; Question answering; Inference; Task (project management); Language model; Code (set theory); Visual reasoning","score_opus":0.009377410152993777,"score_gpt":0.29137206559817114,"score_spread":0.2819946554451774,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7093322906","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37176415,0.018467866,0.2629467,0.0057304567,0.0027182635,0.0032883543,0.12369681,0.15540047,0.055986963],"genre_scores_gemma":[0.47251904,0.002158601,0.24274866,0.0020994735,0.0003141778,0.0019837513,0.26183674,0.0049900785,0.011349392],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9958169,0.0019062166,0.0003141242,0.0010552538,0.0006373901,0.00027022779],"domain_scores_gemma":[0.9901564,0.0066020926,0.00024518336,0.0016020549,0.0010000748,0.0003941882],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005976774,0.0030831671,0.0010332745,0.0020872264,0.001102836,0.0024958018,0.004457594,0.0034187054,0.010669399],"category_scores_gemma":[0.024689253,0.00068259786,0.0024846427,0.0017223515,0.0012246083,0.003865291,0.003091249,0.0035156403,0.005078654],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020369587,0.0022753333,0.011245304,0.003993182,0.0011153892,0.0005747991,0.000604094,0.37246296,0.0057198894,0.010435088,0.23849633,0.35104066],"study_design_scores_gemma":[0.00059380266,0.00066007365,0.0025353741,0.00020717933,0.00015384724,0.00025328912,0.0004057355,0.9465889,0.0065036,0.011291421,0.030714888,0.00009200622],"about_ca_topic_score_codex":0.036061384,"about_ca_topic_score_gemma":0.052350745,"teacher_disagreement_score":0.036061384,"about_ca_system_score_codex":0.002906004,"about_ca_system_score_gemma":0.0029190872,"threshold_uncertainty_score":0.07170296},"labels":[],"label_agreement":null},{"id":"W7103154917","doi":"10.48550/arxiv.2510.25801","title":"Metis-SPECS: Decoupling Multimodal Learning via Self-distilled Preference-based Cold Start","year":2025,"lang":"","type":"preprint","venue":"Open MIND","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Generalization; Reinforcement learning; Task (project management); Task analysis; Verifiable secret sharing; Memorization","score_opus":0.0642883970235779,"score_gpt":0.3301872525248995,"score_spread":0.2658988555013216,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7103154917","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03743405,0.0005396732,0.9487197,0.00040958734,0.00010758995,0.00016946658,0.0003304632,0.008705429,0.003584123],"genre_scores_gemma":[0.66712713,0.00024964247,0.31865326,0.001404233,0.00009508526,0.0006368184,0.0019664948,0.001191682,0.00867568],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998461,0.00047227752,0.00008468575,0.0005228169,0.00028015694,0.00017908409],"domain_scores_gemma":[0.9971432,0.0013139118,0.00016758846,0.0007052966,0.00045151552,0.00021847944],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002403985,0.0023216344,0.0018232882,0.0007151002,0.00077743526,0.001408256,0.004665705,0.0019925921,0.005350088],"category_scores_gemma":[0.0077260095,0.001086602,0.0014998706,0.0007577231,0.0017499344,0.0038837958,0.0049718297,0.004812116,0.0020801725],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008229799,0.00053902867,0.00376494,0.00043779798,0.00023278655,0.00025348135,0.00043177992,0.49249938,0.015985953,0.019731265,0.011658707,0.45364192],"study_design_scores_gemma":[0.000039923118,0.00015186126,0.0001797518,0.000019625459,0.00001810812,0.000032046435,0.000025032781,0.981174,0.0038147946,0.013372397,0.0011520995,0.00002041293],"about_ca_topic_score_codex":0.0043063113,"about_ca_topic_score_gemma":0.0076485877,"teacher_disagreement_score":0.005350088,"about_ca_system_score_codex":0.0013310442,"about_ca_system_score_gemma":0.002195137,"threshold_uncertainty_score":0.017897785},"labels":[],"label_agreement":null},{"id":"W7103204388","doi":"","title":"Metis-SPECS: Decoupling Multimodal Learning via Self-distilled Preference-based Cold Start","year":2025,"lang":"","type":"article","venue":"ArXiv.org","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Generalization; Reinforcement learning; Task (project management); Task analysis; Verifiable secret sharing; Memorization","score_opus":0.036303223380937735,"score_gpt":0.28117547225993805,"score_spread":0.24487224887900033,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7103204388","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03743405,0.0005396732,0.9487197,0.00040958734,0.00010758995,0.00016946658,0.0003304632,0.008705429,0.003584123],"genre_scores_gemma":[0.66712713,0.00024964247,0.31865326,0.001404233,0.00009508526,0.0006368184,0.0019664948,0.001191682,0.00867568],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998461,0.00047227752,0.00008468575,0.0005228169,0.00028015694,0.00017908409],"domain_scores_gemma":[0.9971432,0.0013139118,0.00016758846,0.0007052966,0.00045151552,0.00021847944],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002403985,0.0023216344,0.0018232882,0.0007151002,0.00077743526,0.001408256,0.004665705,0.0019925921,0.005350088],"category_scores_gemma":[0.0077260095,0.001086602,0.0014998706,0.0007577231,0.0017499344,0.0038837958,0.0049718297,0.004812116,0.0020801725],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008229799,0.00053902867,0.00376494,0.00043779798,0.00023278655,0.00025348135,0.00043177992,0.49249938,0.015985953,0.019731265,0.011658707,0.45364192],"study_design_scores_gemma":[0.000039923118,0.00015186126,0.0001797518,0.000019625459,0.00001810812,0.000032046435,0.000025032781,0.981174,0.0038147946,0.013372397,0.0011520995,0.00002041293],"about_ca_topic_score_codex":0.0043063113,"about_ca_topic_score_gemma":0.0076485877,"teacher_disagreement_score":0.005350088,"about_ca_system_score_codex":0.0013310442,"about_ca_system_score_gemma":0.002195137,"threshold_uncertainty_score":0.017897785},"labels":[],"label_agreement":null},{"id":"W7103328463","doi":"","title":"ABC-CNN: An Attention Based Convolutional Neural Network for Visual Question Answering","year":2016,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Question answering; Convolutional neural network; Focus (optics); Feature (linguistics); Natural language; Task (project management); Benchmark (surveying); Image (mathematics); Deep learning","score_opus":0.01630660674486519,"score_gpt":0.31374695009455467,"score_spread":0.2974403433496895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7103328463","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09007564,0.0066432245,0.84584,0.0019612883,0.00073177775,0.00063676434,0.005840529,0.02595849,0.022312311],"genre_scores_gemma":[0.6247534,0.0019573907,0.33374125,0.00207386,0.00018509099,0.00042772424,0.013174829,0.0004211673,0.023265287],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99972206,0.00004071702,0.000012248655,0.00011136973,0.000060465612,0.00005319234],"domain_scores_gemma":[0.9997029,0.00007735284,0.00002567762,0.000057168792,0.000106875574,0.00003010671],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00044028656,0.0014265615,0.00054831535,0.00072455814,0.00031492306,0.0007004888,0.0020233488,0.0014413567,0.0049206847],"category_scores_gemma":[0.0012140152,0.0004013186,0.00077687314,0.00072241254,0.0004790515,0.0017216654,0.0010708638,0.0016454817,0.0017192087],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047044782,0.00043994526,0.0045071403,0.000647924,0.00027985204,0.00034246832,0.00021446087,0.11146164,0.0625156,0.01568475,0.072890595,0.7305452],"study_design_scores_gemma":[0.000032615008,0.000116673815,0.0013046309,0.000042461306,0.00006201909,0.000107871376,0.000029928156,0.96011275,0.015693892,0.009934204,0.012536509,0.000026434162],"about_ca_topic_score_codex":0.02764797,"about_ca_topic_score_gemma":0.036553793,"teacher_disagreement_score":0.02764797,"about_ca_system_score_codex":0.0016620299,"about_ca_system_score_gemma":0.0010682065,"threshold_uncertainty_score":0.05497408},"labels":[],"label_agreement":null},{"id":"W7103895521","doi":"10.1109/tase.2025.3628670","title":"Interactive Semantics-Enhanced Vision-Language Model-Driven Hypergraph Reasoning for Robotic Decision-Making in Proactive Human–Robot Collaboration","year":2025,"lang":"","type":"article","venue":"IEEE Transactions on Automation Science and Engineering","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"National Natural Science Foundation of China","keywords":"Hypergraph; Semantics (computer science); Graph; Process (computing); Construct (python library); Task (project management); Robot; Exploit","score_opus":0.007262317151653083,"score_gpt":0.31891074830319144,"score_spread":0.31164843115153834,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7103895521","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009582494,0.00010980968,0.9881456,0.00022771939,0.000019446717,0.00004653108,0.00009423375,0.0006511282,0.0011230055],"genre_scores_gemma":[0.5707246,0.00027004007,0.42521745,0.00032537978,0.000035519883,0.00020233013,0.0005493084,0.00019293666,0.002482423],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991215,0.00023267049,0.000057706096,0.00031455138,0.00019237438,0.000081332786],"domain_scores_gemma":[0.99898463,0.0005162795,0.00011112139,0.00016322016,0.00015898336,0.00006574796],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00083110796,0.0008631829,0.00058938464,0.00096329226,0.00057530357,0.0012360399,0.0020313244,0.0011231273,0.0022666869],"category_scores_gemma":[0.003282222,0.0004637217,0.0019440831,0.0007280393,0.001037993,0.0035273503,0.0020747022,0.0018437743,0.00033370926],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013475413,0.00018672431,0.0012216355,0.0002774032,0.0001552166,0.00061254215,0.0009258107,0.7451205,0.014492969,0.09336399,0.0030710262,0.14043742],"study_design_scores_gemma":[0.000008092064,0.00001513591,0.00009351499,0.0000072472826,0.000016221094,0.000034632176,0.000044219167,0.96018934,0.0015910942,0.037287083,0.0007027745,0.000010661607],"about_ca_topic_score_codex":0.012958805,"about_ca_topic_score_gemma":0.015551055,"teacher_disagreement_score":0.012958805,"about_ca_system_score_codex":0.0015558018,"about_ca_system_score_gemma":0.001766506,"threshold_uncertainty_score":0.02576673},"labels":[],"label_agreement":null},{"id":"W7106015545","doi":"10.7939/83017","title":"Development of a Question Answering System Over Building Codes using Retrieval Augmented Generation","year":2025,"lang":"en","type":"dissertation","venue":"University of Alberta Library","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Metric (unit); Rank (graph theory); Code (set theory); Adaptation (eye); Scheme (mathematics); Question answering; Key (lock); Data retrieval","score_opus":0.01316818907790685,"score_gpt":0.24095319270402268,"score_spread":0.22778500362611584,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7106015545","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.050287228,0.0006394974,0.79585505,0.0009897926,0.00025821675,0.0018704553,0.0069880653,0.136794,0.006317741],"genre_scores_gemma":[0.16898715,0.00028974543,0.79581475,0.00071819004,0.00008552463,0.0008700125,0.024631333,0.0011293566,0.0074738236],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9985336,0.00032501135,0.00012287899,0.0005175943,0.00035896277,0.00014194999],"domain_scores_gemma":[0.9976749,0.000830758,0.00008750757,0.00039112876,0.0009009128,0.00011474954],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002070672,0.0012473303,0.0009203084,0.0014007611,0.00063682336,0.0012577963,0.002141398,0.0015750224,0.007731897],"category_scores_gemma":[0.005235312,0.00045807398,0.0013680292,0.0008390956,0.0004562413,0.0025115781,0.0014146125,0.001785745,0.0060452577],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005307241,0.0007741951,0.0038648064,0.0010934444,0.00019128757,0.0007373151,0.0011518791,0.026122548,0.09324158,0.0053974623,0.07272165,0.7941732],"study_design_scores_gemma":[0.00024080757,0.00056834426,0.0046427674,0.00010731973,0.00012632301,0.0005309088,0.00079136004,0.8413267,0.07869936,0.0066858116,0.06613658,0.00014372832],"about_ca_topic_score_codex":0.026761204,"about_ca_topic_score_gemma":0.023881676,"teacher_disagreement_score":0.026761204,"about_ca_system_score_codex":0.0012689498,"about_ca_system_score_gemma":0.0022113991,"threshold_uncertainty_score":0.053210855},"labels":[],"label_agreement":null},{"id":"W7115568583","doi":"10.1109/cvprw63382.2024.11301672","title":"Retraction Notice: T2VBench: Benchmarking Temporal Dynamics for Text-to-Video Generation","year":2024,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Benchmarking; Benchmark (surveying); Dynamics (music); Generative grammar; Temporal database; Temporal scales","score_opus":0.032084276902560584,"score_gpt":0.3291478281031487,"score_spread":0.2970635512005881,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7115568583","genre_codex":"methods","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11205535,0.008466326,0.4289454,0.019379733,0.05733198,0.0036156247,0.12992579,0.16205391,0.07822585],"genre_scores_gemma":[0.3291404,0.0020875165,0.30906928,0.0054535167,0.003231717,0.0027250738,0.25910234,0.024354659,0.06483542],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9919436,0.0026580044,0.0006598058,0.0010909114,0.0032799726,0.00036769087],"domain_scores_gemma":[0.9694486,0.0133203035,0.0008823162,0.0036288921,0.01088447,0.0018353973],"candidate_categories":["research_integrity"],"consensus_categories":[],"category_scores_codex":[0.01011242,0.0019309829,0.0009966514,0.0018565147,0.000980289,0.0034232605,0.00244471,0.0021541507,0.021796983],"category_scores_gemma":[0.07521517,0.0003929913,0.00092515256,0.0013167121,0.00079275476,0.0031461706,0.0027652043,0.003006026,0.011151836],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009174799,0.00022163178,0.0035135953,0.0011009119,0.000141912,0.00035261226,0.00039204894,0.017471047,0.009945882,0.0056868806,0.76940787,0.19084814],"study_design_scores_gemma":[0.00081723434,0.002179115,0.011023498,0.0007356874,0.00019232702,0.001194581,0.00075210084,0.30810985,0.04090619,0.0121072335,0.62158614,0.00039612068],"about_ca_topic_score_codex":0.012743952,"about_ca_topic_score_gemma":0.015356427,"teacher_disagreement_score":0.9978458,"about_ca_system_score_codex":0.0019176634,"about_ca_system_score_gemma":0.002369227,"threshold_uncertainty_score":0.07291812},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"not_applicable","genre":"other","about_ca_system":false,"about_ca_topic":false,"confidence":"high"},{"model":"gpt","categories":["research_integrity"],"domain":null,"study_design":"not_applicable","genre":"editorial","about_ca_system":false,"about_ca_topic":false,"confidence":"high"}],"label_agreement":"split"},{"id":"W7116113878","doi":"10.3103/s1060992x25601654","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","year":2025,"lang":"en","type":"article","venue":"Optical Memory and Neural Networks","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Key (lock); Spatial analysis; Visualization; Virtual reality; Data collection","score_opus":0.02165555263727299,"score_gpt":0.25402088083715496,"score_spread":0.23236532819988198,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7116113878","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041546173,0.00031237537,0.94869566,0.00047078857,0.000088978,0.00008314109,0.00070509734,0.005147707,0.0029501386],"genre_scores_gemma":[0.7421361,0.00028057644,0.24935499,0.00029604865,0.00005836608,0.00027686104,0.0016002681,0.0006497917,0.005347014],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996773,0.00009432151,0.000020713878,0.00010801822,0.000058624908,0.000041017312],"domain_scores_gemma":[0.99881244,0.0005839245,0.000107882675,0.00022342429,0.00019163273,0.00008076773],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009718242,0.0011869288,0.00067383464,0.00084764045,0.00033469632,0.0016036127,0.0019522392,0.0010248013,0.004214195],"category_scores_gemma":[0.0048165703,0.000645021,0.0011672771,0.00060076447,0.00054689776,0.0025117944,0.0020299705,0.0019224882,0.0010030924],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021668283,0.00014524137,0.0022976317,0.00009121918,0.000111988775,0.00007339348,0.00021443513,0.8128569,0.003617479,0.007945228,0.0031543754,0.16927543],"study_design_scores_gemma":[0.000006324825,0.00001745238,0.00009492002,0.000006726425,0.0000059897143,0.000007787224,0.0000144649985,0.9946672,0.00060731807,0.00407346,0.00049361715,0.0000047661397],"about_ca_topic_score_codex":0.016699458,"about_ca_topic_score_gemma":0.022355754,"teacher_disagreement_score":0.016699458,"about_ca_system_score_codex":0.0008845659,"about_ca_system_score_gemma":0.0014678442,"threshold_uncertainty_score":0.033204496},"labels":[],"label_agreement":null},{"id":"W7116291721","doi":"10.2139/ssrn.5935496","title":"AI-Generated Visual Content in Education: A Systematic Review of the Past Decade","year":2025,"lang":"","type":"preprint","venue":"SSRN Electronic Journal","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institute for Christian Studies; University of Toronto","funders":"","keywords":"Content (measure theory); Photography; Content analysis; Visual methods; Visualization","score_opus":0.014687450165638914,"score_gpt":0.3194863875845943,"score_spread":0.30479893741895536,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7116291721","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013318623,0.99731207,0.00015998734,0.0003893991,0.00010260801,0.000121030535,0.00022213186,0.000006026468,0.00035490526],"genre_scores_gemma":[0.016410887,0.98121315,0.0010386937,0.0007042125,0.00009670163,0.00027547785,0.00014674256,0.000007542333,0.00010662948],"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","domain_scores_codex":[0.9916604,0.003346745,0.002465432,0.000684598,0.0016510878,0.00019173375],"domain_scores_gemma":[0.9162582,0.07179129,0.006629213,0.0007846713,0.0038726463,0.0006639608],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011701858,0.00087241124,0.004394439,0.011920334,0.0006139521,0.0043155793,0.0016785398,0.0019062416,0.005436451],"category_scores_gemma":[0.063461006,0.00067142246,0.0031910418,0.010254519,0.0017136306,0.0031378963,0.002317483,0.0015207856,0.00040973868],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031909056,0.000100717596,0.0013053417,0.7748225,0.0031529425,0.00008240117,0.0010304279,0.00013872344,0.00021326591,0.00054276385,0.0032917059,0.215],"study_design_scores_gemma":[0.00031258713,0.000324398,0.006811103,0.91802925,0.015660452,0.0003531645,0.0015034834,0.00011323428,0.00026834768,0.00082906213,0.055738863,0.00005600566],"about_ca_topic_score_codex":0.010069375,"about_ca_topic_score_gemma":0.034661055,"teacher_disagreement_score":0.011920334,"about_ca_system_score_codex":0.0035439883,"about_ca_system_score_gemma":0.012380764,"threshold_uncertainty_score":0.061886072},"labels":[],"label_agreement":null},{"id":"W7116666449","doi":"10.1007/978-3-032-13509-4_25","title":"LinguaMark: Do Multimodal Models Speak Fairly? A Benchmark-Based Evaluation","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in social networks","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute","funders":"","keywords":"Benchmark (surveying); Generalization; Key (lock); Code (set theory); Language model; Computational linguistics","score_opus":0.016982273114531085,"score_gpt":0.30576615892638775,"score_spread":0.28878388581185666,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7116666449","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.60539865,0.023176394,0.18257782,0.008643115,0.0044613117,0.003083982,0.033296224,0.038446266,0.100916296],"genre_scores_gemma":[0.8062488,0.0023811695,0.09690236,0.0017998172,0.00068856525,0.0020624588,0.06326433,0.004318848,0.022333752],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98106223,0.013652496,0.0008359187,0.0015317061,0.0023266214,0.00059098675],"domain_scores_gemma":[0.9593594,0.031817485,0.00060929166,0.003780046,0.003161455,0.0012721735],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018856216,0.0035771416,0.0024878192,0.0024385345,0.0019293203,0.004978959,0.0038053077,0.0047941273,0.02129394],"category_scores_gemma":[0.06544055,0.0007290283,0.0015147792,0.0015568525,0.0014671589,0.00840307,0.005393324,0.00341562,0.009721454],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.016708512,0.005525869,0.013981369,0.0041011586,0.0028759248,0.000625041,0.0012532197,0.12251345,0.008552889,0.010540217,0.24803141,0.56529105],"study_design_scores_gemma":[0.0025203817,0.003820951,0.010462864,0.00062660006,0.0011392386,0.0006391262,0.0019150149,0.91632867,0.012137748,0.021074902,0.029013623,0.00032093306],"about_ca_topic_score_codex":0.009004973,"about_ca_topic_score_gemma":0.00979643,"teacher_disagreement_score":0.02129394,"about_ca_system_score_codex":0.0021035993,"about_ca_system_score_gemma":0.0015976242,"threshold_uncertainty_score":0.099722385},"labels":[],"label_agreement":null},{"id":"W7116933240","doi":"10.1109/etncc66224.2025.11299754","title":"Automatic Image Tagging and Captioning Using Transformer-Based Vision-Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Trinity College","funders":"","keywords":"Closed captioning; Transformer; Encoder; Feature extraction; Natural language; Visualization","score_opus":0.012700024091263288,"score_gpt":0.32299977244502204,"score_spread":0.31029974835375873,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7116933240","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015426045,0.00022415229,0.9736302,0.00019645989,0.000089899666,0.00012823427,0.00040677068,0.0076343706,0.0022638345],"genre_scores_gemma":[0.4414716,0.0005778835,0.54532295,0.00038928207,0.00008320162,0.000239114,0.0032564718,0.00096181885,0.007697685],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995454,0.00007957207,0.000030693038,0.00016844715,0.000117851036,0.000057965204],"domain_scores_gemma":[0.9991097,0.00034890627,0.00008237502,0.00013589948,0.0002727484,0.00005032911],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007788564,0.00089110126,0.0005355283,0.0012074779,0.00035394033,0.001544241,0.001449273,0.0007594849,0.0030087365],"category_scores_gemma":[0.0024871458,0.0004743788,0.0015047651,0.00069466257,0.0006348369,0.0024943086,0.0009701239,0.0015113343,0.0029461137],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046109347,0.00029601302,0.002152595,0.0003419048,0.000117012074,0.0003661319,0.00049233984,0.17574003,0.060044367,0.024025487,0.014062516,0.7219006],"study_design_scores_gemma":[0.000011956585,0.00004426083,0.00025317052,0.000014398276,0.000021847309,0.000101623315,0.000063133295,0.96960235,0.018524125,0.008087049,0.0032524748,0.000023586468],"about_ca_topic_score_codex":0.007056862,"about_ca_topic_score_gemma":0.00869004,"teacher_disagreement_score":0.007056862,"about_ca_system_score_codex":0.0013511286,"about_ca_system_score_gemma":0.00097278296,"threshold_uncertainty_score":0.014031589},"labels":[],"label_agreement":null},{"id":"W7117142650","doi":"10.5281/zenodo.17646067","title":"AMVICC Image Results & Evaluations","year":2025,"lang":"","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Centennial College","funders":"","keywords":"Image (mathematics); Relation (database); Documentation; Benchmark (surveying); Image processing; Test (biology)","score_opus":0.03376656010708667,"score_gpt":0.3214794476973359,"score_spread":0.2877128875902492,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7117142650","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08811388,0.010010641,0.09157854,0.0030979947,0.0059026405,0.0034616643,0.1961492,0.25458434,0.34710118],"genre_scores_gemma":[0.19383009,0.001978025,0.1171754,0.0016629897,0.00072683214,0.0018981998,0.5541265,0.045894973,0.08270699],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.990153,0.0017913209,0.0004325709,0.0011848323,0.005552959,0.0008852447],"domain_scores_gemma":[0.9881372,0.0014495524,0.00028613748,0.0025677516,0.0069660014,0.0005933409],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007026416,0.00380948,0.0018866503,0.003961442,0.001386225,0.00471963,0.005299549,0.0029652386,0.047856413],"category_scores_gemma":[0.017905777,0.00070771354,0.0017464088,0.002565036,0.0010331298,0.0023573022,0.0026439442,0.0021419646,0.03704507],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002444969,0.0007270777,0.0011212468,0.0017401846,0.00027142806,0.0002490835,0.0001279043,0.026363574,0.012139982,0.004061704,0.7677975,0.18295537],"study_design_scores_gemma":[0.002262594,0.0019393066,0.01136596,0.0008872139,0.000460495,0.0009233883,0.00040464188,0.20085546,0.099989615,0.010136172,0.6704076,0.00036749718],"about_ca_topic_score_codex":0.01970136,"about_ca_topic_score_gemma":0.014789478,"teacher_disagreement_score":0.047856413,"about_ca_system_score_codex":0.0031763252,"about_ca_system_score_gemma":0.0016798347,"threshold_uncertainty_score":0.16009569},"labels":[],"label_agreement":null},{"id":"W7117145747","doi":"10.5281/zenodo.17646068","title":"AMVICC Image Results & Evaluations","year":2025,"lang":"","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Centennial College","funders":"","keywords":"Image (mathematics); Relation (database); Documentation; Benchmark (surveying); Image processing; Test (biology)","score_opus":0.03376656010708667,"score_gpt":0.3214794476973359,"score_spread":0.2877128875902492,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7117145747","genre_codex":"other","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08811388,0.010010641,0.09157854,0.0030979947,0.0059026405,0.0034616643,0.1961492,0.25458434,0.34710118],"genre_scores_gemma":[0.19383009,0.001978025,0.1171754,0.0016629897,0.00072683214,0.0018981998,0.5541265,0.045894973,0.08270699],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.990153,0.0017913209,0.0004325709,0.0011848323,0.005552959,0.0008852447],"domain_scores_gemma":[0.9881372,0.0014495524,0.00028613748,0.0025677516,0.0069660014,0.0005933409],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007026416,0.00380948,0.0018866503,0.003961442,0.001386225,0.00471963,0.005299549,0.0029652386,0.047856413],"category_scores_gemma":[0.017905777,0.00070771354,0.0017464088,0.002565036,0.0010331298,0.0023573022,0.0026439442,0.0021419646,0.03704507],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002444969,0.0007270777,0.0011212468,0.0017401846,0.00027142806,0.0002490835,0.0001279043,0.026363574,0.012139982,0.004061704,0.7677975,0.18295537],"study_design_scores_gemma":[0.002262594,0.0019393066,0.01136596,0.0008872139,0.000460495,0.0009233883,0.00040464188,0.20085546,0.099989615,0.010136172,0.6704076,0.00036749718],"about_ca_topic_score_codex":0.01970136,"about_ca_topic_score_gemma":0.014789478,"teacher_disagreement_score":0.047856413,"about_ca_system_score_codex":0.0031763252,"about_ca_system_score_gemma":0.0016798347,"threshold_uncertainty_score":0.16009569},"labels":[],"label_agreement":null},{"id":"W7117239628","doi":"10.2139/ssrn.5964393","title":"ELSSA: Explainable Large Language Model-Based Decision Support for Public Transit Incident Management","year":2025,"lang":"","type":"preprint","venue":"SSRN Electronic Journal","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Supervisor; Context (archaeology); Incident management; Ontology; Graph; Incident report; Pipeline (software); Decision support system","score_opus":0.012841728711211984,"score_gpt":0.3009893901624156,"score_spread":0.28814766145120363,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7117239628","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029845122,0.00025768054,0.8403093,0.0013071211,0.00028854256,0.0004017234,0.011380192,0.11372351,0.0024867982],"genre_scores_gemma":[0.34070086,0.0002735587,0.6281203,0.00057673856,0.00014266427,0.0007735112,0.021521408,0.0027693764,0.0051216176],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99898285,0.00044295378,0.000071679104,0.00020159614,0.0002245382,0.00007635763],"domain_scores_gemma":[0.9960477,0.0029923827,0.00018402065,0.00036800778,0.0002761076,0.00013186176],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015239489,0.0015670133,0.0008875457,0.0010417239,0.00058923045,0.0015725992,0.0018591178,0.0011550451,0.017545227],"category_scores_gemma":[0.009107055,0.0005737585,0.0019895306,0.0005396051,0.00046145718,0.0020131662,0.0026619923,0.00258589,0.0040839734],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00207685,0.0010672708,0.006532323,0.0009098952,0.00081885915,0.0011453737,0.0008132988,0.45071638,0.0093430495,0.028294852,0.09392315,0.40435866],"study_design_scores_gemma":[0.0000900538,0.000045040062,0.00023068901,0.000019571686,0.000033025382,0.000024102323,0.000041998133,0.97777015,0.0013819941,0.015229085,0.0051185163,0.000015864713],"about_ca_topic_score_codex":0.008778826,"about_ca_topic_score_gemma":0.013775418,"teacher_disagreement_score":0.017545227,"about_ca_system_score_codex":0.00080825645,"about_ca_system_score_gemma":0.0018043469,"threshold_uncertainty_score":0.0586946},"labels":[],"label_agreement":null},{"id":"W7117850735","doi":"","title":"CountGD++: Generalized Prompting for Open-World Counting","year":2025,"lang":"","type":"article","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Geomechanica (Canada)","funders":"Engineering and Physical Sciences Research Council; University of Oxford","keywords":"Object (grammar); Generalization; Flexibility (engineering); Visualization; Object detection; Annotation; Visual Objects; Code (set theory)","score_opus":0.08297339521387051,"score_gpt":0.25077044132108456,"score_spread":0.16779704610721405,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7117850735","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00572266,0.00019845464,0.9261649,0.00039985348,0.00012399221,0.00022343837,0.002761274,0.06019851,0.00420698],"genre_scores_gemma":[0.10739508,0.00019462989,0.8716484,0.00075823435,0.00008698196,0.001110451,0.0071607097,0.0052058995,0.0064395275],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9972308,0.00082027336,0.00018806916,0.0010479391,0.00053183245,0.00018103517],"domain_scores_gemma":[0.9949032,0.0024711024,0.00028871227,0.0015423835,0.0005546705,0.0002399529],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033467936,0.00243013,0.0014588111,0.0017352963,0.0010006789,0.0032907845,0.0062133134,0.0027493641,0.019348482],"category_scores_gemma":[0.019777466,0.00097361195,0.0017370469,0.0011798604,0.0018372595,0.007782775,0.0077528073,0.003928977,0.0076320954],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00076759054,0.00038750083,0.0042434284,0.0011843015,0.000117352334,0.00037469823,0.0012029366,0.07490568,0.012842968,0.1146267,0.11424144,0.67510533],"study_design_scores_gemma":[0.00010580044,0.000074381496,0.00057248655,0.00012521147,0.000027183465,0.00017704422,0.00018219708,0.7642603,0.01470678,0.1536128,0.06606803,0.00008780521],"about_ca_topic_score_codex":0.0066242414,"about_ca_topic_score_gemma":0.013191829,"teacher_disagreement_score":0.019348482,"about_ca_system_score_codex":0.0021541847,"about_ca_system_score_gemma":0.0021039154,"threshold_uncertainty_score":0.06472707},"labels":[],"label_agreement":null},{"id":"W7124161218","doi":"10.1109/icpads67057.2025.11323053","title":"SwiftReTaKe: Quick and Accurate Redundancy Reduction for Cloud-Edge Collaborative Video-Language Understanding","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"National Key Research and Development Program of China; National Natural Science Foundation of China","keywords":"Cloud computing; Latency (audio); Inference; Redundancy (engineering); Software deployment; Data transmission; Overhead (engineering); Relevance (law); Process (computing)","score_opus":0.0298403845264314,"score_gpt":0.33504563952510236,"score_spread":0.30520525499867096,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124161218","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024111869,0.000691697,0.9584951,0.00027893166,0.00013686872,0.00016124509,0.00043562858,0.014301439,0.0013872525],"genre_scores_gemma":[0.29573914,0.00046878093,0.6957303,0.00040346265,0.000108081505,0.00025219988,0.0020826533,0.001110823,0.004104466],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99922705,0.00009023888,0.00004135521,0.00019255532,0.00034237214,0.00010646864],"domain_scores_gemma":[0.999059,0.00027529447,0.00009686332,0.00027594794,0.00022138956,0.00007154429],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007735533,0.0014198336,0.0011863658,0.0012464521,0.00076319335,0.001201687,0.0028458338,0.0010017172,0.0039253323],"category_scores_gemma":[0.004091653,0.00055871555,0.0008464701,0.00090184924,0.0004904587,0.0031127853,0.0024002735,0.0014252437,0.0020139827],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00068441743,0.0004307791,0.0014408508,0.0002660123,0.00013025415,0.00052116666,0.00041123814,0.05074408,0.06523089,0.0056489357,0.030750018,0.84374124],"study_design_scores_gemma":[0.000055286306,0.00009852184,0.0004265573,0.0000139130525,0.000029172897,0.0001919614,0.000103882616,0.9662319,0.02317902,0.004517304,0.005124588,0.000027926955],"about_ca_topic_score_codex":0.011070876,"about_ca_topic_score_gemma":0.02012535,"teacher_disagreement_score":0.011070876,"about_ca_system_score_codex":0.0006898094,"about_ca_system_score_gemma":0.0018471674,"threshold_uncertainty_score":0.02201289},"labels":[],"label_agreement":null},{"id":"W7124303850","doi":"10.65109/kagr5160","title":"One-Shot Learning from a Demonstration with Hierarchical Latent Language","year":2023,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Microsoft (Canada)","funders":"","keywords":"Generalization; Task (project management); Inference; Principle of compositionality; sort; Suite; Task analysis","score_opus":0.047606030914207946,"score_gpt":0.3014580807053827,"score_spread":0.25385204979117476,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124303850","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1874147,0.0005103253,0.8053023,0.0005848323,0.00008172579,0.00020354307,0.00043043288,0.0024612066,0.0030108679],"genre_scores_gemma":[0.89133286,0.00010866875,0.10359696,0.00018878368,0.000037945632,0.00014818026,0.00086698413,0.000110560075,0.0036090938],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.999385,0.00020348387,0.00003082139,0.00023214886,0.00008887491,0.000059646856],"domain_scores_gemma":[0.9960867,0.0026441056,0.00023792229,0.00059715967,0.00019756549,0.00023645697],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001622122,0.0008695323,0.00087210245,0.0003979679,0.00043177907,0.00085317576,0.0020573803,0.0011916902,0.0032436254],"category_scores_gemma":[0.0075956513,0.00070724724,0.0007610371,0.00032630665,0.0009796433,0.0025128329,0.0020249335,0.0024781688,0.0005551174],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013320058,0.0008097352,0.006442183,0.0005189308,0.00029604975,0.0005596994,0.00083874475,0.6773026,0.01952274,0.022954734,0.007069544,0.26235303],"study_design_scores_gemma":[0.00002974014,0.00013230219,0.00039564696,0.000012455806,0.0000115256935,0.000032292257,0.000040204344,0.9843997,0.002015327,0.012416657,0.0005018314,0.00001228113],"about_ca_topic_score_codex":0.0060815793,"about_ca_topic_score_gemma":0.010840765,"teacher_disagreement_score":0.0060815793,"about_ca_system_score_codex":0.00089597364,"about_ca_system_score_gemma":0.00094665703,"threshold_uncertainty_score":0.012092352},"labels":[],"label_agreement":null},{"id":"W7124309595","doi":"10.65109/yyrj6883","title":"OPEx: A Large Language Model-Powered Framework for Embodied Instruction Following","year":2024,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Microsoft (Canada); Université de Montréal","funders":"","keywords":"Embodied cognition; Bridge (graph theory); Natural language; Language model; Language assessment; Higher education","score_opus":0.01866547196441454,"score_gpt":0.34085330740432207,"score_spread":0.32218783543990753,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124309595","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0030307139,0.00019534476,0.97807515,0.00023008678,0.00007543532,0.00012895037,0.00073987455,0.015809955,0.0017144964],"genre_scores_gemma":[0.13010557,0.00037529357,0.85096174,0.0005137491,0.00009824734,0.0008021648,0.00507294,0.0033415023,0.00872881],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99940574,0.00016858961,0.000036645513,0.00019783173,0.00013741553,0.000053826134],"domain_scores_gemma":[0.9990965,0.00048006407,0.000060761093,0.00018721179,0.000116254974,0.0000591945],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012080612,0.0015929686,0.0009018239,0.0007571425,0.00068023434,0.001787467,0.0036282819,0.0016672577,0.01460319],"category_scores_gemma":[0.0047525107,0.0010514836,0.0022209266,0.0005087286,0.0009794603,0.0030666238,0.0037463456,0.0033670287,0.004717906],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005022025,0.00033638472,0.003019532,0.00070898427,0.00031183456,0.0007450165,0.0005770976,0.37141854,0.011694616,0.0822639,0.037643693,0.4907782],"study_design_scores_gemma":[0.000030943927,0.000045877325,0.00017020493,0.00003457471,0.000019848794,0.000076384116,0.00003875335,0.95310944,0.0017666313,0.034448437,0.010243221,0.000015637179],"about_ca_topic_score_codex":0.006100857,"about_ca_topic_score_gemma":0.0137018375,"teacher_disagreement_score":0.01460319,"about_ca_system_score_codex":0.0009342418,"about_ca_system_score_gemma":0.0019647635,"threshold_uncertainty_score":0.048852563},"labels":[],"label_agreement":null},{"id":"W7124958874","doi":"10.59275/j.melba.2025-5ce1","title":"The Trauma THOMPSON Dataset for Real-World Emergency AI","year":2025,"lang":"en","type":"article","venue":"The Journal of Machine Learning for Biomedical Imaging","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"U.S. Army Medical Research and Development Command; National Science Foundation","keywords":"Action (physics); Benchmark (surveying); Object (grammar); Psychological intervention; Emergency response; Foundation (evidence); Visualization","score_opus":0.013844783570730974,"score_gpt":0.3552164647173537,"score_spread":0.3413716811466227,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124958874","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013785415,0.0034845949,0.0062121423,0.0019085091,0.0010758543,0.00071077864,0.94094306,0.013642369,0.018237302],"genre_scores_gemma":[0.009135186,0.00036808723,0.006589737,0.00032654323,0.00006857302,0.0003016652,0.981065,0.0002258114,0.0019192982],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9982431,0.00031607557,0.00022265749,0.0004472449,0.0005858355,0.00018515303],"domain_scores_gemma":[0.99832577,0.0005014327,0.00011367026,0.00041007376,0.0004453576,0.00020376875],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010168547,0.0036046354,0.0014469334,0.0032697066,0.0011525606,0.002047667,0.0044560055,0.0027081862,0.018042438],"category_scores_gemma":[0.0055684173,0.00046316284,0.0018454549,0.0036783745,0.0006269795,0.0019248676,0.0024018018,0.0024315536,0.021499349],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017111699,0.0002669713,0.0017231498,0.0010704905,0.00010680599,0.00021145165,0.000071886425,0.00470097,0.00071485265,0.0008779138,0.9635647,0.02651975],"study_design_scores_gemma":[0.0004957896,0.00028046442,0.01407629,0.0009594468,0.00017126994,0.0011876847,0.00076803623,0.07255171,0.0050577573,0.007323959,0.8969161,0.00021140647],"about_ca_topic_score_codex":0.03718197,"about_ca_topic_score_gemma":0.086002,"teacher_disagreement_score":0.03718197,"about_ca_system_score_codex":0.0021731085,"about_ca_system_score_gemma":0.001940977,"threshold_uncertainty_score":0.0739311},"labels":[],"label_agreement":null},{"id":"W7125605555","doi":"10.1109/cascon66301.2025.00103","title":"A Hybrid XAI Pipeline for Multimodal Video Understanding: From Transformers to LLMS","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Robustness (evolution); Scalability; Transformer; Software deployment; Pipeline (software); Exploit; Interoperability","score_opus":0.032083133484990374,"score_gpt":0.31484030074691716,"score_spread":0.2827571672619268,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7125605555","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0071358206,0.00029019843,0.81971574,0.0003662502,0.000080085825,0.00021009007,0.002218854,0.16696554,0.0030174179],"genre_scores_gemma":[0.17303458,0.00054902176,0.7937429,0.00038712996,0.00010867018,0.00043864892,0.011605799,0.010251561,0.009881775],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9993191,0.00011373367,0.00003992393,0.00025185436,0.00018956921,0.00008586108],"domain_scores_gemma":[0.998735,0.0004453452,0.00007608662,0.00039056042,0.00026533348,0.00008757616],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017088173,0.0020072104,0.0010048671,0.0022753756,0.00060066115,0.003293474,0.0030127517,0.0010867334,0.022460282],"category_scores_gemma":[0.0057942304,0.0007452993,0.002333365,0.0011670131,0.00066666136,0.0045899185,0.0038095561,0.0026568044,0.010426831],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009255694,0.00024764007,0.003499883,0.00044067786,0.0001861401,0.00036924944,0.0007759744,0.021848189,0.029661017,0.02272264,0.049891695,0.8694312],"study_design_scores_gemma":[0.00006414911,0.00018393031,0.0011343085,0.00007373533,0.00010081482,0.000252889,0.0003802894,0.84601706,0.065126285,0.03737304,0.04920545,0.000088020344],"about_ca_topic_score_codex":0.0076841894,"about_ca_topic_score_gemma":0.006865972,"teacher_disagreement_score":0.022460282,"about_ca_system_score_codex":0.0015386655,"about_ca_system_score_gemma":0.0013233143,"threshold_uncertainty_score":0.07513714},"labels":[],"label_agreement":null},{"id":"W7125938337","doi":"10.1109/smc58881.2025.11342935","title":"SkyNet: An Extensible Edge-Cloud Collaborative Framework for Robots in Long-Horizon Tasks","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Robot; Executable; Modular design; Set (abstract data type); Commodity","score_opus":0.016238226415821375,"score_gpt":0.3441080191156529,"score_spread":0.3278697926998315,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7125938337","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021546334,0.00027500984,0.9345647,0.0003095961,0.00014058161,0.00035670964,0.0007234884,0.036318183,0.0057654325],"genre_scores_gemma":[0.3647753,0.00027393206,0.6236218,0.00035035325,0.000046350484,0.00081960036,0.0021020123,0.002195363,0.0058152806],"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","domain_scores_codex":[0.99951696,0.00010973964,0.000032830874,0.00014120321,0.00011515607,0.000084054],"domain_scores_gemma":[0.9994118,0.00021752225,0.00004549654,0.00013601164,0.000068447196,0.00012058079],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010114767,0.0012027508,0.00064148306,0.0004195027,0.00073418097,0.0008919695,0.0031437727,0.0010845967,0.007497857],"category_scores_gemma":[0.0023604825,0.0005300016,0.00092685275,0.00037363957,0.0007979711,0.0023701412,0.0029277494,0.0013520875,0.0016410806],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010474623,0.0006619455,0.0029625345,0.00048374315,0.00018627207,0.00059477426,0.0006510283,0.73427147,0.012586015,0.031923443,0.029132916,0.18549839],"study_design_scores_gemma":[0.00008300228,0.000065038425,0.0002301796,0.000011379811,0.000012698827,0.00003951094,0.00005039636,0.9775333,0.0019886342,0.01141017,0.008553495,0.000022253365],"about_ca_topic_score_codex":0.016349282,"about_ca_topic_score_gemma":0.028765809,"teacher_disagreement_score":0.016349282,"about_ca_system_score_codex":0.00086618034,"about_ca_system_score_gemma":0.0024599126,"threshold_uncertainty_score":0.032508254},"labels":[],"label_agreement":null},{"id":"W7125957124","doi":"10.1109/smc58881.2025.11343454","title":"Grounded Multi-modal Conversation for Zero-shot Visual Question Answering","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Question answering; Conversation; Comprehension; Modalities; Semantics (computer science); Natural language; Focus (optics); Bridging (networking)","score_opus":0.03046320982700265,"score_gpt":0.3708238761415693,"score_spread":0.3403606663145667,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7125957124","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02988501,0.002096939,0.9429682,0.0009358391,0.0001543143,0.00061805267,0.0011026957,0.018105837,0.0041331006],"genre_scores_gemma":[0.48485225,0.0005933625,0.5012689,0.0017246768,0.00020898496,0.0009129167,0.005156337,0.0007474716,0.0045350953],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9938606,0.0030160872,0.00023321167,0.0016929344,0.0008218098,0.00037539177],"domain_scores_gemma":[0.99512476,0.0031610446,0.00019134353,0.0006585748,0.00059055886,0.00027375284],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038524438,0.0022816446,0.0015975033,0.0011686166,0.0012261255,0.0020818876,0.004732974,0.004097177,0.008016054],"category_scores_gemma":[0.013873874,0.00064755563,0.00215042,0.0006224885,0.0013994391,0.0060277027,0.0069641415,0.003396036,0.0035570727],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023960378,0.0013940283,0.004769375,0.0027670444,0.00058339775,0.0011186857,0.008793117,0.074600615,0.10862089,0.03419996,0.04384483,0.71691203],"study_design_scores_gemma":[0.00015213384,0.0005462101,0.0012162951,0.00011938038,0.00013774079,0.0006609808,0.0018390291,0.85831136,0.02669413,0.08493565,0.025238503,0.00014856021],"about_ca_topic_score_codex":0.0074629034,"about_ca_topic_score_gemma":0.008206902,"teacher_disagreement_score":0.008016054,"about_ca_system_score_codex":0.0013612184,"about_ca_system_score_gemma":0.001729906,"threshold_uncertainty_score":0.026816368},"labels":[],"label_agreement":null},{"id":"W7126109747","doi":"10.1109/bibm66473.2025.11356692","title":"Recurrent Visual Feature Extraction and Stereo Attentions for CT Report Generation","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Feature extraction; Feature (linguistics); Pattern recognition (psychology); Visualization; Encoding (memory); Computed tomography; Benchmark (surveying); Image (mathematics)","score_opus":0.02757592389559764,"score_gpt":0.38309097058611785,"score_spread":0.3555150466905202,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7126109747","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022319918,0.00065210415,0.96536523,0.00033267,0.00010222263,0.00013031327,0.00053787185,0.009341674,0.0012179081],"genre_scores_gemma":[0.5644327,0.00058216025,0.42306438,0.0006830595,0.00024790593,0.00032404801,0.003256475,0.0007003908,0.006708891],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99920386,0.0001428228,0.000042574786,0.00028476067,0.00020756588,0.000118451695],"domain_scores_gemma":[0.9990854,0.0003514981,0.00012495201,0.00017254053,0.00020611618,0.00005940342],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008189986,0.0015990996,0.0012275962,0.0015115013,0.0003200382,0.00079330185,0.0026913583,0.0010916595,0.002887937],"category_scores_gemma":[0.002895379,0.0005447904,0.0018619407,0.0011117609,0.00042307132,0.0013820858,0.001011047,0.0012547002,0.0013247391],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031685593,0.00024396995,0.0009957254,0.00018924373,0.0001051694,0.00040859342,0.00015732246,0.08964036,0.039537564,0.0041004363,0.01229018,0.85201454],"study_design_scores_gemma":[0.000022441196,0.00007891508,0.00040132608,0.000006599833,0.000035392368,0.00014592054,0.000020955631,0.98034656,0.013744174,0.0035689266,0.0016108551,0.000017898405],"about_ca_topic_score_codex":0.008637071,"about_ca_topic_score_gemma":0.009947083,"teacher_disagreement_score":0.008637071,"about_ca_system_score_codex":0.0011866309,"about_ca_system_score_gemma":0.0010509319,"threshold_uncertainty_score":0.017173588},"labels":[],"label_agreement":null},{"id":"W7126165022","doi":"10.21428/594757db.b52553ec","title":"Taxonomic Reasoning for Rare Arthropods: Combining Dense Image Captioning and RAG for Interpretable Classification","year":2025,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Context (archaeology); Biodiversity; Closed captioning; Taxonomic rank; Matching (statistics); Interpretability; Contextual image classification; Convolutional neural network","score_opus":0.01547139791379805,"score_gpt":0.30211488575979084,"score_spread":0.2866434878459928,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7126165022","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06867969,0.0019468232,0.8964963,0.002099189,0.00037117646,0.0003506301,0.002182524,0.02054086,0.0073328638],"genre_scores_gemma":[0.4316138,0.0006686199,0.5580639,0.0012072315,0.00024537742,0.00017605607,0.004524822,0.00060310564,0.0028970172],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993724,0.00016478213,0.00003500627,0.00021916341,0.00013592094,0.000072671144],"domain_scores_gemma":[0.99822897,0.0007654638,0.00019229294,0.00039195147,0.00033333528,0.00008807541],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013187464,0.0010512142,0.0004785239,0.0021736978,0.0004142808,0.001618391,0.0016365672,0.0014502025,0.0036494185],"category_scores_gemma":[0.0055178944,0.0002952959,0.0011640647,0.00081712654,0.00080196443,0.0028638956,0.0015637864,0.0014923072,0.0018855454],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044080685,0.00025204069,0.004046271,0.00047829328,0.00009542867,0.00071722275,0.00062493206,0.05357539,0.06285578,0.009415222,0.021563431,0.8459353],"study_design_scores_gemma":[0.00002890413,0.00012665361,0.0022559904,0.0001106756,0.000065905646,0.00038937567,0.00027611136,0.9250223,0.027817536,0.029751925,0.014102724,0.000052020485],"about_ca_topic_score_codex":0.004046194,"about_ca_topic_score_gemma":0.0051920055,"teacher_disagreement_score":0.004046194,"about_ca_system_score_codex":0.0008175319,"about_ca_system_score_gemma":0.00067621813,"threshold_uncertainty_score":0.012208521},"labels":[],"label_agreement":null},{"id":"W7126258716","doi":"10.1145/3785987.3786101","title":"Automating Classroom Observation: AI-Enabled Behavior Monitoring for Adaptive Educational Management","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Robustness (evolution); Natural language understanding; Perception; Recall; Adaptive learning; Transformer; On the fly","score_opus":0.03428215026433837,"score_gpt":0.34611824780998585,"score_spread":0.31183609754564745,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7126258716","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08657653,0.0020705718,0.8749588,0.0009196316,0.00018707699,0.00029044435,0.00434209,0.022402829,0.008252048],"genre_scores_gemma":[0.6440334,0.0009042978,0.3430835,0.00044564085,0.00008284382,0.00037526648,0.004754892,0.00041437874,0.0059058145],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994586,0.00014073441,0.000019892559,0.00021377139,0.000102938946,0.000064097236],"domain_scores_gemma":[0.99935514,0.0002273438,0.00009646579,0.00012338422,0.000115791736,0.00008178275],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005221467,0.00079902296,0.00038371736,0.0010994546,0.00022120727,0.00074013066,0.00094169046,0.0006866591,0.002296537],"category_scores_gemma":[0.0022947667,0.00021495127,0.0004040123,0.0004953872,0.0003181457,0.0011020032,0.0013440433,0.0010907312,0.0016299908],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033681534,0.00057748705,0.02929047,0.0005597693,0.000113299124,0.00024121431,0.0008204639,0.017518982,0.07331892,0.0031266997,0.022085482,0.8520105],"study_design_scores_gemma":[0.000058922236,0.0003453099,0.04380868,0.00023039001,0.00011844993,0.00036458572,0.0011218567,0.8204311,0.066996105,0.023702817,0.04270739,0.00011434122],"about_ca_topic_score_codex":0.005072374,"about_ca_topic_score_gemma":0.013139703,"teacher_disagreement_score":0.005072374,"about_ca_system_score_codex":0.0005250771,"about_ca_system_score_gemma":0.0009624596,"threshold_uncertainty_score":0.010085702},"labels":[],"label_agreement":null},{"id":"W7126429335","doi":"10.18653/v1/2024.findings-eacl.14","title":"An Examination of the Robustness of Reference-Free Image Captioning Evaluation Metrics","year":2024,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Canadian Institute for Advanced Research; Nvidia","keywords":"Closed captioning; Robustness (evolution); Image (mathematics); Image processing; Pattern recognition (psychology)","score_opus":0.05109654786901989,"score_gpt":0.3427500987957064,"score_spread":0.2916535509266865,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7126429335","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5080059,0.038024083,0.29155797,0.002699162,0.00529411,0.003651125,0.04073592,0.067971334,0.042060405],"genre_scores_gemma":[0.70247966,0.0018924724,0.21312098,0.0011117597,0.0006249058,0.0013675999,0.06821676,0.0056557786,0.005530045],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.97076696,0.012147431,0.003250158,0.005244668,0.0075763557,0.0010144734],"domain_scores_gemma":[0.90935737,0.049001906,0.0052862787,0.015307305,0.019273136,0.001773991],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.027231991,0.0047749276,0.0018051725,0.009712998,0.0017925397,0.0051924856,0.0035752095,0.003048148,0.0036230073],"category_scores_gemma":[0.11702716,0.0005915578,0.0014199412,0.0049181283,0.001857677,0.0055262083,0.004789618,0.003138187,0.002472574],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0037484877,0.00093122415,0.03931122,0.005986211,0.0025565682,0.00051727524,0.0017829427,0.07600989,0.030752527,0.0071383333,0.14907013,0.6821952],"study_design_scores_gemma":[0.0005705359,0.0038807986,0.080300495,0.0016522068,0.0010127182,0.002538264,0.002419646,0.67813945,0.11343559,0.021582184,0.09359042,0.0008777478],"about_ca_topic_score_codex":0.009654256,"about_ca_topic_score_gemma":0.009569305,"teacher_disagreement_score":0.972768,"about_ca_system_score_codex":0.0030300035,"about_ca_system_score_gemma":0.001442686,"threshold_uncertainty_score":0.14401823},"labels":[],"label_agreement":null},{"id":"W7126440079","doi":"10.21428/594757db.dd360009","title":"Multi-modal News Understanding with Professionally LabelledVideos (ReutersViLNews)","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Thomson Reuters (Canada); Vector Institute; University of British Columbia","funders":"","keywords":"Event (particle physics); Context (archaeology); Subject (documents); Action (physics); Benchmark (surveying); Domain (mathematical analysis); Big data","score_opus":0.04817114733350052,"score_gpt":0.31169287511279403,"score_spread":0.2635217277792935,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7126440079","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.067187,0.0066778213,0.05814309,0.0017499296,0.0013421252,0.0017600933,0.7717604,0.059162665,0.032216858],"genre_scores_gemma":[0.03948475,0.00066484645,0.05983684,0.000340157,0.00021811527,0.00063243887,0.8929505,0.0009910403,0.004881308],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99695873,0.000622477,0.00024604794,0.0009972053,0.00078520883,0.00039040713],"domain_scores_gemma":[0.9966467,0.0011137599,0.00030359224,0.0009820053,0.00072908256,0.0002248781],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019190602,0.0036466073,0.0012223881,0.010178349,0.0013134741,0.0029062792,0.0029411928,0.0039887438,0.009555816],"category_scores_gemma":[0.008030897,0.0006261118,0.0021604102,0.0047073606,0.00071458414,0.0049531655,0.00345956,0.0027221995,0.011307648],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001480251,0.0011256174,0.0071392204,0.0043753153,0.0005785469,0.0013240987,0.0013921807,0.0100417845,0.018558905,0.003956442,0.65546304,0.29456466],"study_design_scores_gemma":[0.00043311287,0.0006363533,0.036647104,0.0013562978,0.00049469597,0.0019480212,0.0041844123,0.18167725,0.03506161,0.009602566,0.7275054,0.00045316783],"about_ca_topic_score_codex":0.037357684,"about_ca_topic_score_gemma":0.065560095,"teacher_disagreement_score":0.037357684,"about_ca_system_score_codex":0.0020925058,"about_ca_system_score_gemma":0.0013294966,"threshold_uncertainty_score":0.0742805},"labels":[],"label_agreement":null},{"id":"W7130595512","doi":"10.1109/fllm67465.2025.11391205","title":"Multi-modal Causal RAG for Aviation Accident Analysis and Risk Prediction","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Aviation; Accident (philosophy); Pipeline (software); Topic model; Aviation accident; Aviation safety; Upload; Causal analysis","score_opus":0.010943431764660986,"score_gpt":0.30692521164738124,"score_spread":0.29598177988272023,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7130595512","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019904155,0.0007546872,0.9709553,0.00073271303,0.00003611434,0.00013317508,0.0011204145,0.0043541095,0.002009305],"genre_scores_gemma":[0.5949563,0.00052545045,0.39966565,0.00025686237,0.00006969606,0.00016406724,0.0022551557,0.00016752159,0.0019393865],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993191,0.00028701374,0.000033967506,0.0001819337,0.00013807406,0.00004000019],"domain_scores_gemma":[0.99820685,0.0012295288,0.00020090006,0.0001808017,0.00013444564,0.000047449485],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013065428,0.00072349556,0.0003280409,0.0025800867,0.00042340095,0.000895099,0.0010574955,0.0008272396,0.0051197223],"category_scores_gemma":[0.004914191,0.0002864794,0.0013079883,0.0010844686,0.00062821765,0.0015565642,0.0012520977,0.0010075484,0.0006510793],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005159184,0.0004186862,0.009879153,0.000826663,0.00036100045,0.0016253412,0.0014210013,0.3152961,0.017422568,0.054295294,0.014249793,0.5836884],"study_design_scores_gemma":[0.000023429606,0.00005713021,0.0017981045,0.000057918194,0.000080663376,0.00024795602,0.0002262789,0.9320602,0.004555369,0.0539853,0.006867738,0.00004002223],"about_ca_topic_score_codex":0.0060807867,"about_ca_topic_score_gemma":0.009624453,"teacher_disagreement_score":0.0060807867,"about_ca_system_score_codex":0.00087973784,"about_ca_system_score_gemma":0.0008394284,"threshold_uncertainty_score":0.017127156},"labels":[],"label_agreement":null},{"id":"W7131068881","doi":"10.1109/iccvw69036.2025.00121","title":"MedVisionLlama: Leveraging Pre-Trained Large Language Model Layers to Enhance Medical Image Segmentation","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Montreal Neurological Institute and Hospital","funders":"","keywords":"Segmentation; Image segmentation; Encoder; Scale-space segmentation; Segmentation-based object categorization; Feature (linguistics); Generalization; Scalability","score_opus":0.006658952244160694,"score_gpt":0.35728470193251877,"score_spread":0.3506257496883581,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7131068881","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027817097,0.003053134,0.9227198,0.00096386,0.00036079125,0.0001785468,0.0018107052,0.039835874,0.0032603198],"genre_scores_gemma":[0.32278433,0.0013523494,0.6474017,0.002181523,0.0002215766,0.0004619489,0.006488613,0.0042056856,0.014902318],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999617,0.000097953576,0.00001978059,0.00014230105,0.00007190589,0.000051152023],"domain_scores_gemma":[0.9994288,0.00029423175,0.00004281061,0.00010055137,0.000085607135,0.000047975336],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001068149,0.0012071064,0.00077619497,0.0006835278,0.0003037966,0.0014143593,0.0018205557,0.0015235359,0.005123586],"category_scores_gemma":[0.003984353,0.0007542033,0.0013657035,0.00047658946,0.0005594453,0.0013559227,0.0017094137,0.0021818401,0.0035308427],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007314436,0.00021335516,0.0020960937,0.00054994144,0.0004100992,0.00038184735,0.00024748687,0.25150433,0.06703695,0.0070995525,0.040862415,0.62886655],"study_design_scores_gemma":[0.000048213744,0.00012805055,0.00041897641,0.00004793225,0.00005851192,0.00020163169,0.00002631535,0.96124566,0.022169929,0.0065341005,0.009080239,0.000040433817],"about_ca_topic_score_codex":0.0069780317,"about_ca_topic_score_gemma":0.013811632,"teacher_disagreement_score":0.0069780317,"about_ca_system_score_codex":0.00095380796,"about_ca_system_score_gemma":0.0012466285,"threshold_uncertainty_score":0.01714009},"labels":[],"label_agreement":null},{"id":"W7131081286","doi":"10.1109/iccvw69036.2025.00476","title":"A Survey on Vision-Language-Action Models for Autonomous Driving","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"TRACE (psycholinguistics); Control (management); Measure (data warehouse); Natural (archaeology); Natural language","score_opus":0.03708258920698129,"score_gpt":0.3680725928216169,"score_spread":0.3309900036146356,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7131081286","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012661066,0.10233894,0.83998823,0.0053138393,0.000835889,0.00033703505,0.005691867,0.0057742107,0.027058978],"genre_scores_gemma":[0.325331,0.17203347,0.45111424,0.0022838593,0.0012999914,0.0012808217,0.028233357,0.0020156256,0.01640768],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99892247,0.00025244852,0.00012299616,0.00025265972,0.00037545475,0.00007393305],"domain_scores_gemma":[0.9979962,0.0012388593,0.00008760791,0.00027780922,0.0003382955,0.00006129016],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014424702,0.0017222442,0.0011573447,0.0018903275,0.00052849855,0.0025344857,0.0029944726,0.0016305326,0.007289278],"category_scores_gemma":[0.0059776325,0.0009798887,0.00203301,0.0022542258,0.0007129884,0.0038009696,0.0015842437,0.0021287359,0.0027194964],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015542538,0.00023728874,0.004935011,0.003906345,0.0003448046,0.00028792472,0.0004819299,0.25991184,0.0021380973,0.13085328,0.044770185,0.55197793],"study_design_scores_gemma":[0.00002673865,0.00012946062,0.002052029,0.0009529433,0.00016721764,0.00030193245,0.00020789288,0.60445905,0.0019295033,0.15422486,0.2354478,0.00010068308],"about_ca_topic_score_codex":0.01635189,"about_ca_topic_score_gemma":0.015483703,"teacher_disagreement_score":0.01635189,"about_ca_system_score_codex":0.0019368053,"about_ca_system_score_gemma":0.00273128,"threshold_uncertainty_score":0.03251338},"labels":[],"label_agreement":null},{"id":"W7131093438","doi":"10.1109/iccvw69036.2025.00221","title":"Refining Naive Annotations with Limited Expert Guidance for Semantic Segmentation: A Case Study on Underwater Echograms","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Fisheries and Oceans Canada; Canadian Water Network; ASL Environmental Sciences (Canada); University of Victoria","funders":"Alliance de recherche numérique du Canada","keywords":"Metadata; Annotation; Ground truth; Context (archaeology); Intersection (aeronautics); Artificial neural network; Segmentation; Convolutional neural network; Domain (mathematical analysis)","score_opus":0.042033903670718034,"score_gpt":0.3603349623937063,"score_spread":0.3183010587229883,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7131093438","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40865025,0.0010984408,0.5673692,0.0010965333,0.00017553369,0.0003139747,0.001875139,0.012560519,0.006860353],"genre_scores_gemma":[0.55470866,0.0002916839,0.43319434,0.00037954873,0.000058761747,0.00013970623,0.004888013,0.0016376853,0.004701629],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9967957,0.0011805983,0.00017735704,0.0011020759,0.00052668457,0.00021756717],"domain_scores_gemma":[0.98966086,0.007374596,0.00026160764,0.001305247,0.0011573324,0.00024035238],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004372505,0.0018230877,0.001082918,0.0016728617,0.0011416336,0.001752974,0.0019940273,0.0028524515,0.001910917],"category_scores_gemma":[0.011429848,0.00053556485,0.00094392453,0.0012846518,0.0016075707,0.0023609474,0.0019434559,0.0018209533,0.001501634],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019978562,0.00059964525,0.012711863,0.0011993472,0.0002491397,0.003221902,0.004648995,0.3122662,0.086185805,0.008052985,0.025802564,0.5430636],"study_design_scores_gemma":[0.000094772586,0.00023513126,0.004522539,0.00011728236,0.0000910922,0.00071337697,0.0013789816,0.9060962,0.057447545,0.00997643,0.01924766,0.00007896751],"about_ca_topic_score_codex":0.014418252,"about_ca_topic_score_gemma":0.035607852,"teacher_disagreement_score":0.014418252,"about_ca_system_score_codex":0.0013718172,"about_ca_system_score_gemma":0.0012371872,"threshold_uncertainty_score":0.028668642},"labels":[],"label_agreement":null},{"id":"W7131095609","doi":"10.1109/iccvw69036.2025.00782","title":"DIVE-Doc: Downscaling Foundational Image Visual Encoder into Hierarchical Architecture for DocVQA","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Encoder; Distillation; Image (mathematics); Code (set theory); Architecture; Forcing (mathematics); Visualization","score_opus":0.008780846628456837,"score_gpt":0.339472735249167,"score_spread":0.33069188862071014,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7131095609","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027010065,0.0013625569,0.8234573,0.000596178,0.00042991096,0.00031952022,0.0030891043,0.12827158,0.015463739],"genre_scores_gemma":[0.3841789,0.0006668626,0.5641341,0.0013773297,0.00012250346,0.0004827029,0.008817108,0.0071239877,0.03309651],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99972683,0.00003576785,0.000015402085,0.00009757293,0.000081953964,0.000042408992],"domain_scores_gemma":[0.99953187,0.000122064645,0.000016268115,0.00016670121,0.00012452986,0.0000385649],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00059496885,0.0009530808,0.000453651,0.00045829453,0.00042832523,0.0013877453,0.0023017544,0.0009837558,0.015140239],"category_scores_gemma":[0.002411419,0.00042437235,0.00068699615,0.0003760768,0.00076754246,0.002229133,0.0018094986,0.0021082126,0.006220732],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078595884,0.00021176886,0.0026322394,0.00067254127,0.00015554707,0.00032593164,0.00058724405,0.13655207,0.045718186,0.03898239,0.13386346,0.6395126],"study_design_scores_gemma":[0.00010006255,0.00016111703,0.0004011551,0.000065755674,0.000035607016,0.00012845173,0.00009648532,0.8916757,0.029756851,0.017949877,0.05957937,0.000049611048],"about_ca_topic_score_codex":0.014168476,"about_ca_topic_score_gemma":0.031114299,"teacher_disagreement_score":0.015140239,"about_ca_system_score_codex":0.0012956262,"about_ca_system_score_gemma":0.0013057453,"threshold_uncertainty_score":0.050649107},"labels":[],"label_agreement":null},{"id":"W7133067082","doi":"","title":"&quot;Flobject&quot; Analysis: Learning about Static Images from Motion","year":2011,"lang":"en","type":"dissertation","venue":"TSpace","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Generalization; Unsupervised learning; Motion (physics); Pattern recognition (psychology); Representation (politics); Image (mathematics); Object (grammar); Pyramid (geometry); Field (mathematics)","score_opus":0.013653236677320619,"score_gpt":0.32381778625437724,"score_spread":0.31016454957705664,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7133067082","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016511174,0.0010531571,0.97491044,0.0010759074,0.00013162737,0.000094741576,0.0006858905,0.0015137164,0.0040233866],"genre_scores_gemma":[0.44626537,0.0023839104,0.52413464,0.000668779,0.0004197115,0.00023425795,0.003514265,0.0007316259,0.021647384],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99938667,0.000108140994,0.000026536723,0.00024661163,0.00016130718,0.00007066653],"domain_scores_gemma":[0.99903464,0.00032957247,0.00013059037,0.00027025118,0.00018714987,0.000047789275],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010089581,0.0007681315,0.0010857639,0.001504584,0.0006578571,0.001512607,0.001548705,0.0011705317,0.004842665],"category_scores_gemma":[0.003791274,0.00044228422,0.0009624,0.0019249502,0.001256791,0.0031768177,0.0009154607,0.0012640161,0.0021778254],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025863008,0.00006826408,0.001699268,0.00017199355,0.00007370346,0.00013502968,0.00024241475,0.045923095,0.017471872,0.023574255,0.019595409,0.89078605],"study_design_scores_gemma":[0.0000146697475,0.00007571158,0.0038710725,0.00004450655,0.00003592418,0.00021422634,0.000116636234,0.9154536,0.016725162,0.046472766,0.016925318,0.000050397954],"about_ca_topic_score_codex":0.013641224,"about_ca_topic_score_gemma":0.013462189,"teacher_disagreement_score":0.013641224,"about_ca_system_score_codex":0.0016005054,"about_ca_system_score_gemma":0.0008365091,"threshold_uncertainty_score":0.02712369},"labels":[],"label_agreement":null},{"id":"W7134188119","doi":"10.1109/bigdata66926.2025.11400804","title":"The Hybrid Deployment Architecture for Explainable and Robust Video Understanding","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Software deployment; Architecture; Key (lock); Systems architecture; Domain (mathematical analysis); Robustness (evolution)","score_opus":0.031838450459745475,"score_gpt":0.2813186936731368,"score_spread":0.2494802432133913,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7134188119","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011694221,0.00014338836,0.96999973,0.00028177936,0.000044175296,0.00006913597,0.00025331922,0.013913312,0.0036009105],"genre_scores_gemma":[0.3826374,0.0003729518,0.59462845,0.00032970656,0.00007048589,0.00023767819,0.002037966,0.0019283192,0.017757056],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996679,0.00005646114,0.0000151700715,0.000115667906,0.0000961075,0.000048833783],"domain_scores_gemma":[0.9994981,0.00011608053,0.000025964679,0.0001717589,0.0001468633,0.00004113943],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00063853076,0.000913481,0.00048031737,0.0006221131,0.0003675489,0.0011112933,0.0019420765,0.0014403111,0.012209443],"category_scores_gemma":[0.0021161716,0.0005112421,0.0005479309,0.00053911004,0.0006848775,0.003046214,0.002519095,0.0014669598,0.0044909995],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055983686,0.00024072487,0.002819424,0.00026784922,0.00012317885,0.0005826056,0.0008161433,0.08112534,0.08161982,0.061181817,0.036222808,0.7344405],"study_design_scores_gemma":[0.000021282573,0.00008028865,0.0008040489,0.00003190733,0.000035149453,0.00019718005,0.00011514491,0.9307273,0.025987204,0.023019588,0.018952671,0.000028261267],"about_ca_topic_score_codex":0.005995224,"about_ca_topic_score_gemma":0.008365322,"teacher_disagreement_score":0.012209443,"about_ca_system_score_codex":0.0006151256,"about_ca_system_score_gemma":0.000789342,"threshold_uncertainty_score":0.04084468},"labels":[],"label_agreement":null},{"id":"W7137029536","doi":"10.1109/itsc60802.2025.11423805","title":"Adver-City: Open-Source Multi-Modal Dataset for Collaborative Perception Under Adverse Weather Conditions","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Perception; Adverse weather; Climate change; Nowcasting","score_opus":0.039672793922927586,"score_gpt":0.37323132760810146,"score_spread":0.33355853368517385,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7137029536","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01971371,0.0015571584,0.013086161,0.0005234771,0.0005838498,0.0004491869,0.92100847,0.035156507,0.0079213455],"genre_scores_gemma":[0.01731964,0.00022695209,0.014014067,0.00017145085,0.000039966013,0.00031463985,0.96552306,0.00086431,0.0015260074],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99859077,0.00020470373,0.000103422884,0.0005567065,0.00032335983,0.00022098738],"domain_scores_gemma":[0.99898213,0.0001802654,0.00007473272,0.00037202938,0.00026325416,0.00012746843],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007981996,0.004198578,0.0014104936,0.002322747,0.0012788457,0.0020319866,0.0049115075,0.003071292,0.013025614],"category_scores_gemma":[0.0033415854,0.0008414801,0.002645071,0.002774069,0.0007429106,0.0022653586,0.003333203,0.002608657,0.018621635],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000770393,0.0004827474,0.005961496,0.0018599809,0.00037158112,0.00034407547,0.0002712276,0.010208274,0.004199772,0.0011970986,0.92586803,0.048465244],"study_design_scores_gemma":[0.00076770916,0.00044014704,0.045065865,0.0011365456,0.00034918656,0.001502077,0.0016216841,0.12042603,0.015066112,0.0075054984,0.8056039,0.00051521166],"about_ca_topic_score_codex":0.047676913,"about_ca_topic_score_gemma":0.10971979,"teacher_disagreement_score":0.047676913,"about_ca_system_score_codex":0.0016218997,"about_ca_system_score_gemma":0.0015384994,"threshold_uncertainty_score":0.0947988},"labels":[],"label_agreement":null},{"id":"W7143420289","doi":"10.47363/jaicc/icaicc/2025(4)17","title":"The Wisdom of Fusion: In-depth Analysis and Future Outlook of Visual Multimodal Technologies","year":2025,"lang":"","type":"article","venue":"Journal of Artificial Intelligence & Cloud Computing","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Focus (optics); Variety (cybernetics); Presentation (obstetrics); Cognitive neuroscience of visual object recognition; Segmentation; Deep learning; Multimodality; Multimodal learning","score_opus":0.015573289620855942,"score_gpt":0.32937919495661067,"score_spread":0.3138059053357547,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7143420289","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009435958,0.6756052,0.09602808,0.14814688,0.005141796,0.000048626007,0.00030414277,0.00032638843,0.06496287],"genre_scores_gemma":[0.2932843,0.5949714,0.060549412,0.01867184,0.015809841,0.0001364054,0.00035085555,0.00021817863,0.016007723],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99882704,0.00047022602,0.000048168687,0.00014192401,0.00039137102,0.000121218196],"domain_scores_gemma":[0.9950434,0.0031025729,0.0001638632,0.00039549614,0.0010311163,0.00026361257],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0047470084,0.0007038705,0.00078348507,0.002440664,0.0012615853,0.0071792295,0.0014814545,0.0032954563,0.0060867113],"category_scores_gemma":[0.0068252576,0.00036268134,0.0006196234,0.0018495022,0.005390511,0.016896727,0.0032145143,0.0043472727,0.001193758],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010049258,0.00004819719,0.0010198904,0.0010135409,0.000057812806,0.00016323382,0.0011793806,0.0022787128,0.0010237953,0.5887284,0.042188324,0.3621982],"study_design_scores_gemma":[0.000011035144,0.0000690487,0.0010633143,0.0016722587,0.000034114757,0.00044739395,0.0018595231,0.010595098,0.0011062584,0.7254263,0.2576331,0.00008248048],"about_ca_topic_score_codex":0.0016870399,"about_ca_topic_score_gemma":0.0017291452,"teacher_disagreement_score":0.0071792295,"about_ca_system_score_codex":0.002527381,"about_ca_system_score_gemma":0.001657052,"threshold_uncertainty_score":0.02510488},"labels":[],"label_agreement":null},{"id":"W7154561671","doi":"10.1109/imc-ssgp67001.2025.11473981","title":"4LM: Local Lightweight LLMs for Image Captioning on Embedded Systems","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Closed captioning; Image (mathematics); Image processing; Key (lock); Component (thermodynamics)","score_opus":0.011642461619575056,"score_gpt":0.30496486969592274,"score_spread":0.2933224080763477,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7154561671","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014754932,0.000728202,0.8749975,0.00040561304,0.00018742651,0.0002084646,0.00086919504,0.10360946,0.0042391894],"genre_scores_gemma":[0.44023803,0.00063794776,0.536259,0.001019061,0.0001053392,0.00046832004,0.0037938193,0.007415481,0.010062988],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9993755,0.00013783004,0.000043452947,0.00013416052,0.00023351265,0.000075478114],"domain_scores_gemma":[0.9987097,0.00046789984,0.00008132314,0.00043597943,0.00022044324,0.00008467001],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010038437,0.0012846659,0.00066003716,0.0005160519,0.00044407844,0.0014730877,0.0027476822,0.0014014408,0.013086185],"category_scores_gemma":[0.005524285,0.0005553185,0.00090521836,0.0003802677,0.0006985269,0.0036153656,0.0030202186,0.0017376986,0.005081886],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014869154,0.00046214642,0.0024419446,0.0010223495,0.00021953795,0.00081541686,0.000581439,0.19078808,0.0902069,0.026654469,0.0853762,0.5999446],"study_design_scores_gemma":[0.000046584682,0.00012755496,0.00024481877,0.000037715712,0.000023478193,0.000108854525,0.00006165279,0.9390787,0.03416964,0.010433663,0.015619108,0.00004818175],"about_ca_topic_score_codex":0.003689402,"about_ca_topic_score_gemma":0.0060896343,"teacher_disagreement_score":0.013086185,"about_ca_system_score_codex":0.0009247036,"about_ca_system_score_gemma":0.00093181664,"threshold_uncertainty_score":0.043777585},"labels":[],"label_agreement":null},{"id":"W7160014773","doi":"10.1109/iccv51701.2025.00237","title":"FDPT: Federated Discrete Prompt Tuning for Black-Box Visual-Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Beijing Municipal Commission of Education; National Natural Science Foundation of China","keywords":"Key (lock); Identification (biology); Feature (linguistics); Set (abstract data type); Term (time)","score_opus":0.015915098965143006,"score_gpt":0.3358294414760369,"score_spread":0.3199143425108939,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7160014773","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019469397,0.00023600836,0.83321446,0.00024987335,0.0003532878,0.00018045379,0.00076082646,0.14043915,0.0050965855],"genre_scores_gemma":[0.6421652,0.00011284964,0.33660653,0.0006038082,0.00010311348,0.00035826451,0.0015599724,0.008445662,0.01004461],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99909604,0.0001723363,0.00004254207,0.00033651412,0.00021107333,0.00014158394],"domain_scores_gemma":[0.99853384,0.0007556598,0.00005397545,0.00032713785,0.00020915926,0.00012028315],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012138365,0.0017473724,0.0010861751,0.00057654694,0.0005731291,0.0017280205,0.0031617621,0.0017955379,0.03621909],"category_scores_gemma":[0.0070484406,0.0006664174,0.0008244012,0.00034249728,0.0006291158,0.0022251839,0.0026067644,0.0025541016,0.006857982],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0039102957,0.00078289123,0.0023240298,0.0004917674,0.00019463501,0.00049003196,0.00031887233,0.2397344,0.03730459,0.014425482,0.049475927,0.6505471],"study_design_scores_gemma":[0.00012429671,0.00007542279,0.0001257547,0.0000148563295,0.000015221086,0.000034242647,0.000025581017,0.97725064,0.010654114,0.0078345435,0.0038246454,0.000020727879],"about_ca_topic_score_codex":0.005960577,"about_ca_topic_score_gemma":0.0072590103,"teacher_disagreement_score":0.03621909,"about_ca_system_score_codex":0.0010408895,"about_ca_system_score_gemma":0.0015807556,"threshold_uncertainty_score":0.12116492},"labels":[],"label_agreement":null},{"id":"W7160028952","doi":"10.1109/iccv51701.2025.00749","title":"OV-SCAN: Semantically Consistent Alignment for Novel Object Discovery in Open-Vocabulary 3D Object Detection","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Object (grammar); Object detection; Feature (linguistics); Pattern recognition (psychology); Viola–Jones object detection framework","score_opus":0.020977677567380806,"score_gpt":0.31158699577160637,"score_spread":0.2906093182042256,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7160028952","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01434924,0.00081023626,0.9228413,0.00018669729,0.0002704263,0.00026643233,0.0033712937,0.055394255,0.0025100259],"genre_scores_gemma":[0.109406374,0.00040018404,0.8648912,0.00031591702,0.00014494009,0.0003829337,0.015942419,0.0058155456,0.002700449],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9960484,0.0005003065,0.0002719065,0.0015327851,0.0012793753,0.00036722975],"domain_scores_gemma":[0.9966169,0.00095789187,0.00027487936,0.00120527,0.00077905913,0.00016597175],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022453205,0.003026275,0.0032261577,0.0059350627,0.0017985102,0.0035298162,0.0049227397,0.0035109213,0.014763765],"category_scores_gemma":[0.010302872,0.0014325066,0.0024332416,0.005910523,0.0015652798,0.005366438,0.00926133,0.0027303253,0.010324932],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017873434,0.00047220144,0.0035919356,0.0010637408,0.000513547,0.00081175147,0.00066260336,0.011980503,0.06878545,0.020429324,0.06417131,0.8257302],"study_design_scores_gemma":[0.00033787807,0.0005682802,0.0027619402,0.00020210861,0.00033252587,0.0016874003,0.0010916485,0.71535385,0.09426723,0.12516755,0.058029648,0.00019999131],"about_ca_topic_score_codex":0.0053946055,"about_ca_topic_score_gemma":0.011068141,"teacher_disagreement_score":0.014763765,"about_ca_system_score_codex":0.0007817887,"about_ca_system_score_gemma":0.0024048244,"threshold_uncertainty_score":0.04938972},"labels":[],"label_agreement":null},{"id":"W7160031978","doi":"10.1109/iccv51701.2025.00592","title":"Mamba-3VL: Taming State Space Model for 3D Vision Language Learning","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"State (computer science); Space (punctuation); Feature (linguistics); Natural language; State space; Key (lock)","score_opus":0.009304892235748189,"score_gpt":0.32582369203578365,"score_spread":0.31651879980003544,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7160031978","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0049588396,0.000085655236,0.98732644,0.00012114938,0.00006728846,0.000028082524,0.00019457753,0.0065401862,0.0006777422],"genre_scores_gemma":[0.4950234,0.00022579542,0.49358973,0.00052068406,0.000052898675,0.00043945498,0.0016665702,0.0013627858,0.0071187364],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99966836,0.000093272756,0.000014539226,0.00007986788,0.00010416815,0.00003979155],"domain_scores_gemma":[0.9994942,0.00019695371,0.00002800931,0.00012867262,0.000118484975,0.00003373419],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00073268695,0.00062879355,0.0007737849,0.0002968201,0.00042034517,0.0008215111,0.0016887898,0.0012785273,0.0056677717],"category_scores_gemma":[0.0025167058,0.00042924387,0.0007782451,0.0003153867,0.00049538224,0.0010559623,0.0018425375,0.002158385,0.0019849732],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00068484637,0.00023774889,0.0014993756,0.00027527378,0.00018077876,0.00014602233,0.0002495094,0.56622124,0.03281318,0.043240245,0.014829121,0.33962274],"study_design_scores_gemma":[0.000008854922,0.000030585416,0.000057560024,0.0000047272842,0.000005209707,0.0000087168255,0.0000057481334,0.98985493,0.0034179457,0.0049544047,0.0016441022,0.000007205045],"about_ca_topic_score_codex":0.007415295,"about_ca_topic_score_gemma":0.010858702,"teacher_disagreement_score":0.007415295,"about_ca_system_score_codex":0.0005526093,"about_ca_system_score_gemma":0.0010965702,"threshold_uncertainty_score":0.018960595},"labels":[],"label_agreement":null},{"id":"W7160043273","doi":"10.1109/iccv51701.2025.02161","title":"DC-TTA: Divide-and-Conquer Framework for Test-Time Adaptation of Interactive Segmentation","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Ministry of Science and ICT, South Korea; National Research Foundation of Korea","keywords":"Segmentation; Adaptation (eye); Feature (linguistics); Pattern recognition (psychology); Image segmentation","score_opus":0.01516478191265817,"score_gpt":0.3258266459429268,"score_spread":0.31066186403026863,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7160043273","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002962597,0.00019054108,0.9792477,0.00007402852,0.00005121999,0.000092044,0.00017573565,0.016318955,0.0008872445],"genre_scores_gemma":[0.121350616,0.0001802674,0.86763084,0.00028620823,0.00012273043,0.00044467192,0.0011804369,0.0044524437,0.004351892],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985643,0.00028788403,0.000076635326,0.0004320168,0.0004063957,0.00023277126],"domain_scores_gemma":[0.9981681,0.00066601473,0.00010253804,0.00046375755,0.00042743032,0.00017212206],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001980348,0.0018826006,0.0018397841,0.0018997362,0.0010446844,0.0019211208,0.005117564,0.0027783555,0.014287788],"category_scores_gemma":[0.0057150745,0.00087174773,0.0016776498,0.001875682,0.0011041594,0.0017842147,0.0028575435,0.003394752,0.004295619],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008679556,0.00027461274,0.0012096866,0.00018971087,0.00017837416,0.00019371622,0.00027903102,0.11047307,0.029225873,0.01088187,0.01942587,0.82680017],"study_design_scores_gemma":[0.00003150056,0.000055679666,0.00024014297,0.000011093554,0.000024354495,0.000059473172,0.00003747272,0.9780927,0.009192453,0.0074213683,0.0048118657,0.000021946375],"about_ca_topic_score_codex":0.017487276,"about_ca_topic_score_gemma":0.02907913,"teacher_disagreement_score":0.017487276,"about_ca_system_score_codex":0.0014224687,"about_ca_system_score_gemma":0.0029791442,"threshold_uncertainty_score":0.04779744},"labels":[],"label_agreement":null},{"id":"W7160112519","doi":"10.1109/iccv51701.2025.00687","title":"Physics Context Builders: A Modular Framework for Physical Reasoning in Vision-Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Canadian Institute for Advanced Research","keywords":"Context (archaeology); Modular design; Automated reasoning; Physical system; Knowledge representation and reasoning","score_opus":0.016075970502646243,"score_gpt":0.35030521424483335,"score_spread":0.3342292437421871,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7160112519","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0016528157,0.000080179096,0.988608,0.00019007527,0.00004374462,0.00007750885,0.0003921991,0.0073057916,0.0016497426],"genre_scores_gemma":[0.14031303,0.00032051213,0.8503685,0.00037555755,0.00010904105,0.0003812337,0.0015718769,0.0020376323,0.0045226207],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99896324,0.00020271249,0.00008322866,0.0002499069,0.0004055744,0.00009523866],"domain_scores_gemma":[0.99853873,0.0005509946,0.00008455991,0.0005199702,0.00020090018,0.00010482803],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018259708,0.0015229167,0.0013603716,0.0016137732,0.0011967551,0.0043211393,0.005959105,0.0020169578,0.014705968],"category_scores_gemma":[0.006587525,0.0017618856,0.0040614703,0.0010440064,0.0016127072,0.007365177,0.005593844,0.0046379967,0.004705676],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035502177,0.00038357402,0.0016307974,0.0005781532,0.00038707064,0.0005286276,0.0009252208,0.13727278,0.009866848,0.5484906,0.025029346,0.274552],"study_design_scores_gemma":[0.000033507808,0.000024158302,0.00015955082,0.00005853537,0.00009086494,0.000057845027,0.00006419712,0.6135421,0.0058073783,0.36576554,0.0143505065,0.000045836096],"about_ca_topic_score_codex":0.007245014,"about_ca_topic_score_gemma":0.014681847,"teacher_disagreement_score":0.014705968,"about_ca_system_score_codex":0.0011337877,"about_ca_system_score_gemma":0.0019166799,"threshold_uncertainty_score":0.049196303},"labels":[],"label_agreement":null},{"id":"W7160147334","doi":"10.1109/iccv51701.2025.00072","title":"PRISM: Reducing Spurious Implicit Biases in Vision-Language Models with LLM-Guided Embedding Projection","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Spurious relationship; Projection (relational algebra); Embedding; Noise (video); Computation","score_opus":0.022203687657010328,"score_gpt":0.35456226915590394,"score_spread":0.33235858149889363,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7160147334","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013883003,0.00049359,0.9719024,0.00042723503,0.000173887,0.00008400386,0.0003795257,0.01128141,0.0013749647],"genre_scores_gemma":[0.45774704,0.0004401396,0.52517486,0.0012218751,0.00025984974,0.00038768892,0.0025980303,0.0023486395,0.009821873],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99842286,0.0005974698,0.00007550367,0.0003948343,0.00033012033,0.00017919733],"domain_scores_gemma":[0.9967733,0.0015961996,0.00014586118,0.0007491507,0.0005600391,0.00017542078],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020814727,0.0023020117,0.002082383,0.00073203124,0.0007500194,0.0018763301,0.0032003485,0.0030351935,0.008071351],"category_scores_gemma":[0.012243148,0.001224764,0.0013008508,0.00096780766,0.0011502205,0.004244716,0.0051728543,0.0048963293,0.004415208],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011145374,0.0004659319,0.0015586177,0.000611511,0.00033724113,0.00031106887,0.0003088398,0.24900983,0.020334734,0.022968525,0.032103457,0.6708758],"study_design_scores_gemma":[0.000043158783,0.000064616,0.00009060021,0.000017067376,0.000023823957,0.000044220207,0.000022173313,0.9798248,0.0040783025,0.014800076,0.0009765363,0.000014675245],"about_ca_topic_score_codex":0.0076324125,"about_ca_topic_score_gemma":0.012299894,"teacher_disagreement_score":0.008071351,"about_ca_system_score_codex":0.0007968373,"about_ca_system_score_gemma":0.0024415082,"threshold_uncertainty_score":0.02700138},"labels":[],"label_agreement":null},{"id":"W7160163144","doi":"10.1109/iccv51701.2025.01747","title":"LOCATEdit: Graph Laplacian Optimized Cross Attention for Localized Text-Guided Image Editing","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Graph; Image (mathematics); Laplacian matrix; Laplace operator; Image editing; Pattern recognition (psychology)","score_opus":0.0174265668069709,"score_gpt":0.3409412574415373,"score_spread":0.3235146906345664,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7160163144","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014897922,0.00047830507,0.9482236,0.00021717059,0.00028070746,0.00014728436,0.0006386853,0.031307552,0.003808721],"genre_scores_gemma":[0.27650678,0.000382813,0.6793239,0.0010042507,0.00028899068,0.00038936653,0.0038895013,0.0066357907,0.031578626],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995747,0.00006381616,0.00001411207,0.00013717488,0.00013725065,0.00007295643],"domain_scores_gemma":[0.9992465,0.00029986867,0.000036906556,0.00014690883,0.00018799676,0.00008182612],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00058291416,0.0017761523,0.0014158282,0.0012129563,0.0007032808,0.0010957406,0.0028322064,0.002033457,0.018115006],"category_scores_gemma":[0.0023941237,0.0005271384,0.0009767123,0.0011093583,0.00052764185,0.0013393092,0.0023561353,0.0020336243,0.005120689],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00065888086,0.00048918126,0.00045909823,0.00030660594,0.00019145272,0.00032851618,0.00019155264,0.09314382,0.056172457,0.007516261,0.061550718,0.7789914],"study_design_scores_gemma":[0.00003632756,0.000091104,0.00019634484,0.000009307809,0.00002339385,0.000080062106,0.000032584565,0.9784041,0.013114291,0.0045480533,0.0034442774,0.000020169758],"about_ca_topic_score_codex":0.015870344,"about_ca_topic_score_gemma":0.033018347,"teacher_disagreement_score":0.018115006,"about_ca_system_score_codex":0.00075784913,"about_ca_system_score_gemma":0.0011344446,"threshold_uncertainty_score":0.060600758},"labels":[],"label_agreement":null}]}