{"meta":{"query_hash":"29852c204d50","filters":{"topic":"Reinforcement Learning in Robotics"},"cohort_total":1145,"direct_labels_cover":2,"predictions_cover":1145,"exported":1145,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/29852c204d50","api":"https://metacan.xera.ac/api/v1/cohort?topic=Reinforcement+Learning+in+Robotics"},"results":[{"id":"W102912931","doi":"","title":"Inter-Layer Learning Towards Emergent Cooperative Behavior","year":2000,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Testbed; Reinforcement learning; Layer (electronics); Heuristics; Artificial intelligence; Abstraction; Architecture; Process (computing); Simple (philosophy); Abstraction layer; Machine learning; Distributed computing; Online machine learning; Human–computer interaction; Unsupervised learning","score_opus":0.028209157006429064,"score_gpt":0.284066985288857,"score_spread":0.2558578282824279,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W102912931","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10190929,0.00011506974,0.8903664,0.00037226535,0.000029252438,0.00007763261,0.000016366495,0.00075146253,0.0063622897],"genre_scores_gemma":[0.8703487,0.0001023526,0.12653711,0.0001334787,0.000015943262,0.00015344327,0.000037490885,0.000063719795,0.0026077172],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99953234,0.000167949,0.000027741715,0.00008759761,0.000093659546,0.000090674985],"domain_scores_gemma":[0.9986815,0.0005235194,0.00015635083,0.00027992687,0.00023895646,0.000119753546],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013761845,0.000711664,0.00048782714,0.00029817218,0.00047969466,0.0010652752,0.0011806486,0.00075029256,0.0017782036],"category_scores_gemma":[0.0034781878,0.00042992117,0.00041073086,0.00019444976,0.0013833592,0.0016425641,0.0024996046,0.0014174546,0.00043403904],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010513861,0.00018252947,0.0025897298,0.00010124996,0.0000864024,0.00016047359,0.00049092463,0.8523714,0.016064556,0.053883087,0.001167611,0.07279687],"study_design_scores_gemma":[0.000008920342,0.000028522018,0.0001309499,0.0000054790216,0.0000070831697,0.000010364765,0.000025723597,0.9740165,0.0018352714,0.023391005,0.00053596194,0.000004233557],"about_ca_topic_score_codex":0.0014324267,"about_ca_topic_score_gemma":0.0017749158,"teacher_disagreement_score":0.0017782036,"about_ca_system_score_codex":0.0007457127,"about_ca_system_score_gemma":0.0008169572,"threshold_uncertainty_score":0.007278025},"labels":[],"label_agreement":null},{"id":"W1037351197","doi":"10.1613/jair.4676","title":"Approximate Value Iteration with Temporally Extended Actions","year":2015,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; European Commission","keywords":"Landmark; Computer science; Convergence (economics); Bellman equation; Mathematical optimization; Reinforcement learning; Value (mathematics); Function (biology); Term (time); State space; Mathematics; Artificial intelligence; Machine learning; Statistics","score_opus":0.3129026095091614,"score_gpt":0.4403874809011682,"score_spread":0.12748487139200682,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1037351197","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023307381,0.00010963394,0.97524047,0.00007882988,0.000015903513,0.00003171539,0.000023963239,0.00018310429,0.0010089512],"genre_scores_gemma":[0.6523781,0.00012899234,0.34528714,0.00008075738,0.000018404517,0.00019317986,0.00009320434,0.000078354366,0.0017419141],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99902487,0.00043882508,0.000060578543,0.00016352619,0.00022099637,0.000091161965],"domain_scores_gemma":[0.99639976,0.0026177156,0.00030308947,0.0003036343,0.00025695658,0.00011882021],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018194399,0.0007013217,0.00090059725,0.00043142666,0.0003636199,0.00082343246,0.0012103941,0.0010735953,0.0021879298],"category_scores_gemma":[0.009021895,0.00046516428,0.00068642,0.0005604375,0.0015451089,0.0018808014,0.0015289307,0.0014715805,0.00025917825],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001933607,0.000033911187,0.0009379204,0.000059709622,0.00003402476,0.00007814893,0.00015236891,0.9063586,0.0015002812,0.047256634,0.00035015238,0.04304481],"study_design_scores_gemma":[0.000014064965,0.000038539274,0.00005273847,0.000008295465,0.000004173721,0.000015208622,0.0000110135015,0.981024,0.0005599939,0.017954,0.0003122049,0.0000057746784],"about_ca_topic_score_codex":0.0026511108,"about_ca_topic_score_gemma":0.0029556297,"teacher_disagreement_score":0.0026511108,"about_ca_system_score_codex":0.0008666422,"about_ca_system_score_gemma":0.0011035926,"threshold_uncertainty_score":0.009622276},"labels":[],"label_agreement":null},{"id":"W115994101","doi":"","title":"Robust Online Optimization of Reward-Uncertain MDPs","year":2012,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Regret; Computer science; Leverage (statistics); Markov decision process; Minimax; Mathematical optimization; Set (abstract data type); Pruning; Computation; Artificial intelligence; Markov process; Machine learning; Mathematics; Algorithm","score_opus":0.05854918816098995,"score_gpt":0.26757053188534,"score_spread":0.20902134372435008,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W115994101","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037957102,0.00043888934,0.9550579,0.0004567178,0.000037543687,0.000086050866,0.00018006316,0.0005751606,0.0052105063],"genre_scores_gemma":[0.8683033,0.00035632693,0.12759708,0.0001685738,0.000038943883,0.00022567993,0.00027680647,0.00014707279,0.002886244],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985618,0.0005311032,0.00007732256,0.00032409537,0.00028026247,0.0002254095],"domain_scores_gemma":[0.99399096,0.0046397485,0.00055634655,0.00037591928,0.00024209914,0.00019495648],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025060496,0.001713811,0.002365828,0.0005727461,0.0004852835,0.001523806,0.0014671638,0.0017878745,0.002986636],"category_scores_gemma":[0.011567392,0.0010422098,0.00093675667,0.00063633884,0.0016010443,0.001901224,0.0021129076,0.0027186354,0.00039459858],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000054026936,0.000023190913,0.00015751184,0.000045586923,0.00001480201,0.000029528583,0.000017926617,0.98678917,0.00023075851,0.007844491,0.00024137477,0.0045515876],"study_design_scores_gemma":[0.000010823092,0.000014379993,0.000028039014,0.0000066393595,0.0000033196277,0.0000042511433,0.0000032536818,0.9928975,0.0001452706,0.0067812693,0.00010242462,0.0000026960831],"about_ca_topic_score_codex":0.0049861763,"about_ca_topic_score_gemma":0.0043746913,"teacher_disagreement_score":0.0049861763,"about_ca_system_score_codex":0.0020883747,"about_ca_system_score_gemma":0.0020520496,"threshold_uncertainty_score":0.015152216},"labels":[],"label_agreement":null},{"id":"W1178556121","doi":"10.1609/aaai.v29i1.9613","title":"Policy Tree: Adaptive Representation for Policy Gradient","year":2015,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Softmax function; Representation (politics); Computer science; Tree (set theory); Decision tree; Reinforcement learning; Tree structure; Mathematical optimization; Focus (optics); Base (topology); Artificial intelligence; Machine learning; Mathematics; Algorithm; Political science; Binary tree; Law; Artificial neural network","score_opus":0.18186388199228115,"score_gpt":0.36106959719988463,"score_spread":0.17920571520760348,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1178556121","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004235291,0.00017685095,0.9929471,0.00021597263,0.000055404293,0.000068583446,0.00007753186,0.00097609527,0.0012472209],"genre_scores_gemma":[0.39393458,0.0005309758,0.5991588,0.0005081113,0.00012263212,0.00081128813,0.0005536909,0.0006819793,0.0036979336],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99888366,0.00048822036,0.00006932088,0.00020670815,0.0002508586,0.00010123336],"domain_scores_gemma":[0.9968689,0.0020195402,0.00023637695,0.0002987019,0.00043557218,0.00014090302],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026632433,0.0010954087,0.0012882217,0.00094122445,0.0005456338,0.0014492971,0.0019571232,0.002004304,0.0064142602],"category_scores_gemma":[0.014049374,0.0006796563,0.00078053033,0.0008970893,0.0011945883,0.002800903,0.0017435555,0.0030600356,0.001450454],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023344843,0.00013205784,0.0009803402,0.00014897446,0.00004800091,0.00009896752,0.00015835233,0.6915336,0.0021076596,0.07980519,0.007705629,0.21704769],"study_design_scores_gemma":[0.000015936072,0.00002494218,0.00003497278,0.0000136214485,0.0000047827866,0.000015566822,0.0000055196415,0.97838384,0.00044454387,0.019985303,0.0010647052,0.0000063213274],"about_ca_topic_score_codex":0.0030421328,"about_ca_topic_score_gemma":0.0025676899,"teacher_disagreement_score":0.0064142602,"about_ca_system_score_codex":0.0012132261,"about_ca_system_score_gemma":0.0021731053,"threshold_uncertainty_score":0.021457851},"labels":[],"label_agreement":null},{"id":"W1190299041","doi":"10.3233/wia-140298","title":"Learning cooperative behavior for the shout-ahead architecture","year":2014,"lang":"en","type":"article","venue":"Web Intelligence and Agent Systems An International Journal","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Look-ahead; Architecture; Computer science; Computer architecture; History; Archaeology","score_opus":0.037285219547616065,"score_gpt":0.3072938326640247,"score_spread":0.27000861311640867,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1190299041","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.081432074,0.00009344531,0.91143936,0.00026164608,0.000032719097,0.00006462701,0.000017112669,0.00046563946,0.0061934344],"genre_scores_gemma":[0.8627533,0.00007671817,0.13005728,0.00007868825,0.000013620592,0.000121610494,0.000034630735,0.00004288537,0.00682138],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99980813,0.00005148691,0.000009293615,0.000047120866,0.000053041804,0.00003093827],"domain_scores_gemma":[0.9996282,0.00015045198,0.000034664896,0.000056797937,0.0000759234,0.00005391714],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005858643,0.00036270954,0.00035250213,0.00019948589,0.0005796954,0.000463322,0.0009929664,0.0007343499,0.003293967],"category_scores_gemma":[0.001236177,0.00021543796,0.00032350185,0.0001350211,0.00074954424,0.00076852913,0.0009603345,0.0008729168,0.0004073327],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017904975,0.00019931648,0.0014738847,0.000088707,0.0000766301,0.00023339021,0.0003854098,0.8048739,0.016465396,0.07111682,0.0015688688,0.10333862],"study_design_scores_gemma":[0.000009455793,0.000044774733,0.0000892158,0.00000294782,0.0000065689283,0.000016001659,0.000013934977,0.9879473,0.00092735345,0.010476711,0.00046020132,0.000005478285],"about_ca_topic_score_codex":0.0039292644,"about_ca_topic_score_gemma":0.005733048,"teacher_disagreement_score":0.0039292644,"about_ca_system_score_codex":0.0005196881,"about_ca_system_score_gemma":0.0007428766,"threshold_uncertainty_score":0.011019409},"labels":[],"label_agreement":null},{"id":"W13294968","doi":"10.1016/0021-8707(56)90033-8","title":"Toward Off-Policy Learning Control with Function Approximation","year":2010,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":175,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Temporal difference learning; Function approximation; Approximation algorithm; Function (biology); Computer science; Linear approximation; Bellman equation; Optimal control; Extension (predicate logic); Mathematical optimization; Control (management); Artificial intelligence; Mathematics; Reinforcement learning; Artificial neural network; Nonlinear system","score_opus":0.022667319724685036,"score_gpt":0.27060077476957206,"score_spread":0.24793345504488704,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W13294968","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003765274,0.00013775118,0.99314755,0.00014731786,0.00003939305,0.000035269397,0.000013915043,0.0003708324,0.0023426611],"genre_scores_gemma":[0.6216077,0.00033669677,0.3683889,0.0007247206,0.0001374821,0.00045868685,0.00018371313,0.00040238936,0.007759728],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987147,0.0003347325,0.00006135148,0.00025391916,0.00045269355,0.00018267601],"domain_scores_gemma":[0.99645257,0.0023462828,0.0002544181,0.00027997122,0.0005250287,0.00014171322],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002936141,0.001629104,0.0018601932,0.00083826296,0.0005538244,0.00166337,0.0020498752,0.0018601384,0.003300941],"category_scores_gemma":[0.008435267,0.0007002832,0.0009506204,0.00061411684,0.0021890658,0.0015316738,0.0027894713,0.0035542885,0.0007983472],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011535346,0.00010360542,0.00041942674,0.000090931375,0.000032362525,0.000059084065,0.000120164324,0.86691326,0.001411857,0.055068694,0.0014039674,0.07426123],"study_design_scores_gemma":[0.0000071796326,0.000019315163,0.00001368817,0.000006413277,0.0000019035275,0.000005895445,0.0000029091477,0.991289,0.00020047858,0.008099364,0.00035093536,0.0000028056354],"about_ca_topic_score_codex":0.00585327,"about_ca_topic_score_gemma":0.0026254647,"teacher_disagreement_score":0.00585327,"about_ca_system_score_codex":0.0020377375,"about_ca_system_score_gemma":0.0022791037,"threshold_uncertainty_score":0.015527964},"labels":[],"label_agreement":null},{"id":"W136985510","doi":"10.5555/2034396.2034517","title":"Escaping local optima in POMDP planning as inference","year":2011,"lang":"en","type":"article","venue":"Adaptive Agents and Multi-Agents Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Inference; Partially observable Markov decision process; Local optimum; Reinforcement learning; Computer science; Mathematical optimization; Observable; Greedy algorithm; Artificial intelligence; Machine learning; Mathematics; Markov chain; Markov model","score_opus":0.14580520123192156,"score_gpt":0.3207535313869368,"score_spread":0.17494833015501524,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W136985510","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021550706,0.0002570921,0.97554374,0.00025900867,0.000018458872,0.000043187796,0.000020788595,0.00040059924,0.0019063638],"genre_scores_gemma":[0.7663519,0.00024520798,0.23132463,0.00017704384,0.000029458328,0.00020719023,0.000049113347,0.00011187333,0.0015036461],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989944,0.0004414437,0.00005151698,0.0001668119,0.00022754888,0.000118143376],"domain_scores_gemma":[0.9957508,0.003364935,0.0002818339,0.00028025205,0.00019360613,0.0001286334],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029250209,0.00092575926,0.0016256355,0.0006072966,0.000615276,0.00094094564,0.001495914,0.0012747815,0.0013483847],"category_scores_gemma":[0.008670349,0.00079169637,0.0007210283,0.0006155971,0.0026204176,0.0016983374,0.0021339415,0.0024055156,0.00021627666],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000043562562,0.00001877312,0.00025748025,0.000038557653,0.000024732028,0.000037812348,0.000071515504,0.9690899,0.00043094176,0.017356018,0.00019774934,0.012433004],"study_design_scores_gemma":[0.000013223251,0.000014821702,0.000028608783,0.000006205497,0.0000058604446,0.0000052345003,0.000007509139,0.98516244,0.00025693135,0.014368229,0.00012674803,0.000004116733],"about_ca_topic_score_codex":0.0058028074,"about_ca_topic_score_gemma":0.0055414257,"teacher_disagreement_score":0.0058028074,"about_ca_system_score_codex":0.001348704,"about_ca_system_score_gemma":0.001464654,"threshold_uncertainty_score":0.015469193},"labels":[],"label_agreement":null},{"id":"W137325057","doi":"10.1609/aaai.v26i1.8319","title":"Approximate Policy Iteration with Linear Action Models","year":2021,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates - Technology Futures","keywords":"Benchmark (surveying); Computer science; Action (physics); Mathematical optimization; Linear model; Machine learning; Mathematics","score_opus":0.12002286483175004,"score_gpt":0.3180867896487678,"score_spread":0.19806392481701776,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W137325057","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012190749,0.00032723424,0.9851608,0.0002207296,0.000036042427,0.000051707386,0.000051227307,0.0006756426,0.0012859429],"genre_scores_gemma":[0.6353007,0.0002787504,0.35825124,0.0003608423,0.00007999519,0.00045982064,0.00039480734,0.00022041803,0.004653445],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980427,0.0008632961,0.0001032745,0.00038789128,0.00041909324,0.00018365863],"domain_scores_gemma":[0.9928606,0.0056468286,0.00044415132,0.00042368646,0.00044217342,0.00018242178],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003064049,0.0012763245,0.0024172647,0.0007341575,0.00043508483,0.0014461137,0.0023124602,0.0022189706,0.0028295454],"category_scores_gemma":[0.012621276,0.0010622208,0.0007671948,0.0009814098,0.0017496962,0.0024523034,0.001783129,0.0030633237,0.00073472795],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011239167,0.000059263144,0.00039184673,0.00006920936,0.00003638746,0.000034193352,0.000040163097,0.96673757,0.00032451915,0.0071176337,0.00046400295,0.02461289],"study_design_scores_gemma":[0.000008104369,0.00001860051,0.000017580574,0.00000321405,0.0000021015128,0.000004716902,0.0000027940746,0.9954425,0.000154559,0.0042130672,0.00012999072,0.0000027435835],"about_ca_topic_score_codex":0.0073694726,"about_ca_topic_score_gemma":0.006346474,"teacher_disagreement_score":0.0073694726,"about_ca_system_score_codex":0.001608278,"about_ca_system_score_gemma":0.0028605938,"threshold_uncertainty_score":0.016204417},"labels":[],"label_agreement":null},{"id":"W1433129784","doi":"10.1007/978-3-642-29946-9_16","title":"Automatic Construction of Temporally Extended Actions for MDPs Using Bisimulation Metrics","year":2012,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Markov decision process; Reinforcement learning; Computer science; Bisimulation; Metric (unit); Set (abstract data type); Markov process; Process (computing); Q-learning; Performance metric; Artificial intelligence; Theoretical computer science; Mathematical optimization; Algorithm; Mathematics; Programming language","score_opus":0.056064834870291215,"score_gpt":0.3043537203516835,"score_spread":0.2482888854813923,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1433129784","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010048212,0.000070582355,0.98787737,0.000047494428,0.000023743285,0.00008316277,0.000078078156,0.0007393487,0.0010319728],"genre_scores_gemma":[0.17039445,0.00015493133,0.8264566,0.000050152554,0.000022262195,0.0004828405,0.000495542,0.00054013106,0.001403096],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984316,0.0004201998,0.00015018765,0.0003252842,0.0005383584,0.00013436849],"domain_scores_gemma":[0.9955051,0.0028551344,0.00029432357,0.0005073367,0.00060211285,0.00023599909],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020003545,0.0016755785,0.0018623229,0.0016474553,0.00087604555,0.0013292846,0.0019148672,0.0015113263,0.0054206513],"category_scores_gemma":[0.009132884,0.0014916827,0.0022563438,0.0010074525,0.0016423583,0.00275642,0.005075929,0.0030222766,0.00095721363],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023675774,0.00014208912,0.00067588117,0.0004970961,0.00008049885,0.00021956609,0.0003024094,0.6377551,0.014907143,0.14948012,0.0020262795,0.19367704],"study_design_scores_gemma":[0.000018611177,0.00005250212,0.00006720537,0.000034762597,0.000011871484,0.000027725946,0.000025348016,0.9311885,0.0031924727,0.06412603,0.001243031,0.000011892561],"about_ca_topic_score_codex":0.0018355646,"about_ca_topic_score_gemma":0.0033356498,"teacher_disagreement_score":0.0054206513,"about_ca_system_score_codex":0.001691978,"about_ca_system_score_gemma":0.0022009108,"threshold_uncertainty_score":0.018133879},"labels":[],"label_agreement":null},{"id":"W1482535585","doi":"10.1007/978-3-642-41575-3_15","title":"Controller Compilation and Compression for Resource Constrained Applications","year":2013,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Benchmark (surveying); Lookup table; Controller (irrigation); Compiler; Set (abstract data type); Markov decision process; Computation; Table (database); Decision table; Point (geometry); Resource (disambiguation); Distributed computing; Computer engineering; Algorithm; Artificial intelligence; Data mining; Markov process; Programming language","score_opus":0.016047394966606572,"score_gpt":0.24618815952536066,"score_spread":0.2301407645587541,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1482535585","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034835268,0.0024957021,0.9271072,0.0003336136,0.0003297171,0.0001706771,0.00028315655,0.009856343,0.024588289],"genre_scores_gemma":[0.61838704,0.0015566779,0.35893258,0.0002742821,0.00023174974,0.00031955793,0.00081139547,0.0013590443,0.018127622],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99962914,0.000047945756,0.00002832795,0.00006437792,0.00017677789,0.000053415122],"domain_scores_gemma":[0.9993686,0.00027611715,0.000025305322,0.00020266577,0.00010784603,0.000019533318],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00029737724,0.0008748935,0.00055907643,0.0005565879,0.00045580682,0.00085912365,0.0012927861,0.0005056946,0.00848347],"category_scores_gemma":[0.0015679932,0.00039742974,0.00037143024,0.00078448205,0.0005094276,0.0012785126,0.00096200494,0.0012114577,0.0010925713],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004200823,0.00016292762,0.00031477746,0.00041796797,0.00004435504,0.0003064401,0.00012573956,0.158427,0.03293854,0.045197733,0.017176189,0.7444682],"study_design_scores_gemma":[0.00005480904,0.00010166369,0.00039180624,0.00008067094,0.000030559346,0.00022307807,0.00005199306,0.88229895,0.036488112,0.06309006,0.01715612,0.0000321662],"about_ca_topic_score_codex":0.0017455419,"about_ca_topic_score_gemma":0.0026991987,"teacher_disagreement_score":0.00848347,"about_ca_system_score_codex":0.00041820068,"about_ca_system_score_gemma":0.0007032597,"threshold_uncertainty_score":0.028380036},"labels":[],"label_agreement":null},{"id":"W1491129446","doi":"10.1007/978-3-540-39857-8_29","title":"Using MDP Characteristics to Guide Exploration in Reinforcement Learning","year":2003,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Markov decision process; Artificial intelligence; Bellman equation; Machine learning; State space; Computation; Focus (optics); Variance (accounting); Markov process; Mathematical optimization; Algorithm","score_opus":0.044656664918599175,"score_gpt":0.28567957287483425,"score_spread":0.24102290795623507,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1491129446","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021869354,0.00012819156,0.97554183,0.00013390998,0.000027564478,0.000044829976,0.000045377372,0.00031118467,0.0018977593],"genre_scores_gemma":[0.687607,0.00023066466,0.3093768,0.00009960523,0.000034756078,0.00028818927,0.00014014395,0.0002587412,0.001964065],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99954695,0.00015324929,0.00003746092,0.000059975868,0.00016519958,0.000037207767],"domain_scores_gemma":[0.9938207,0.0047074547,0.00045013017,0.00024599687,0.0005851838,0.00019061156],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00168813,0.00081232673,0.000898946,0.0009391892,0.00035366561,0.0007876667,0.0008219551,0.00084032863,0.0021822492],"category_scores_gemma":[0.012363517,0.0007095213,0.00041917912,0.0005901018,0.00077751494,0.0019736728,0.0011867265,0.0016895834,0.00026842474],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000107328495,0.00003898388,0.0009800561,0.00007888596,0.00001843329,0.000052101568,0.0000611246,0.9256065,0.001664543,0.022986755,0.00066835165,0.047737062],"study_design_scores_gemma":[0.0000071892705,0.000018091552,0.000061281375,0.0000058575306,0.0000025099805,0.000012272298,0.000003868627,0.99336123,0.00037299332,0.005947025,0.00020459175,0.0000030062545],"about_ca_topic_score_codex":0.00193916,"about_ca_topic_score_gemma":0.0024612024,"teacher_disagreement_score":0.0021822492,"about_ca_system_score_codex":0.00083336217,"about_ca_system_score_gemma":0.0009269786,"threshold_uncertainty_score":0.0089277625},"labels":[],"label_agreement":null},{"id":"W1493024987","doi":"10.48550/arxiv.1205.2651","title":"Seeing the Forest Despite the Trees: Large Scale Spatial-Temporal Decision Making","year":2012,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Reinforcement learning; Scale (ratio); Action (physics); Spatial planning; Space (punctuation); Point (geometry); Spatial ecology; State space; State (computer science); Artificial intelligence; Operations research; Geography; Environmental planning; Mathematics; Cartography; Algorithm","score_opus":0.04397477088403532,"score_gpt":0.20018948106472031,"score_spread":0.156214710180685,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1493024987","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12747066,0.00046239822,0.8640233,0.0018915272,0.000055502795,0.00005495308,0.00012078829,0.00040234896,0.0055184187],"genre_scores_gemma":[0.8856083,0.00028325059,0.111123435,0.00018283498,0.00003568386,0.000090571695,0.00011578078,0.00006980315,0.0024902446],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992449,0.0003455233,0.000029310606,0.00018739305,0.00010945937,0.00008336733],"domain_scores_gemma":[0.9962029,0.0028737711,0.0003062901,0.00021890776,0.00013049322,0.00026761455],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013913838,0.00052719115,0.0008317347,0.00027050727,0.00067989057,0.0012741946,0.0012450686,0.0014789897,0.0034532805],"category_scores_gemma":[0.0068056067,0.0004441915,0.00064338965,0.0005594469,0.001953666,0.0029107581,0.0016850558,0.0020201653,0.00027965638],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001022836,0.000056703393,0.0009790641,0.00004518177,0.000031681782,0.00014709677,0.00013393964,0.92975444,0.00073341414,0.045325,0.0008791129,0.021812147],"study_design_scores_gemma":[0.000013373117,0.000018210261,0.00020537678,0.0000054643756,0.0000052671503,0.000019322872,0.000027003956,0.9410893,0.0002201124,0.057848945,0.00054080324,0.000006852124],"about_ca_topic_score_codex":0.010631113,"about_ca_topic_score_gemma":0.009358117,"teacher_disagreement_score":0.010631113,"about_ca_system_score_codex":0.0014980729,"about_ca_system_score_gemma":0.0013539696,"threshold_uncertainty_score":0.02113849},"labels":[],"label_agreement":null},{"id":"W1493952344","doi":"10.1007/978-3-540-24677-0_102","title":"Multiple Reinforcement Learning Agents in a Static Environment","year":2004,"lang":"en","type":"book","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Acadia University","funders":"","keywords":"Reinforcement learning; Computer science; Testbed; Reinforcement; Artificial intelligence; Error-driven learning; State (computer science); Learning classifier system; Human–computer interaction; Machine learning; Computer network","score_opus":0.020084878225391827,"score_gpt":0.24644086449333158,"score_spread":0.22635598626793976,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1493952344","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12501036,0.00045719402,0.85135514,0.00047911218,0.00018099732,0.000100910365,0.000051570652,0.0009358241,0.021428877],"genre_scores_gemma":[0.7827398,0.00043716608,0.18650074,0.00011206006,0.000095206866,0.00017921995,0.00008983402,0.000105751475,0.02974022],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997037,0.000058154856,0.0000145459535,0.00007971046,0.0000963104,0.00004754906],"domain_scores_gemma":[0.9994381,0.00022379037,0.00006781613,0.00006331475,0.00008619572,0.00012088368],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00046297678,0.00073319895,0.0008213432,0.0003808941,0.00073051243,0.00075282465,0.0015933379,0.0012069627,0.004282507],"category_scores_gemma":[0.0014916721,0.0004943327,0.00046358004,0.00040748884,0.00079070154,0.0013669227,0.0019587434,0.00087286934,0.0006756422],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023936965,0.0001536018,0.0006061669,0.000090428395,0.00008139181,0.00043254765,0.0001000661,0.8695744,0.009533966,0.02485191,0.0016845784,0.09265153],"study_design_scores_gemma":[0.000031347274,0.00008607509,0.00017242384,0.0000067956744,0.000018039824,0.00006533956,0.000022005916,0.984597,0.0015825392,0.0118711265,0.0015363193,0.000011034583],"about_ca_topic_score_codex":0.0017051842,"about_ca_topic_score_gemma":0.0019630007,"teacher_disagreement_score":0.004282507,"about_ca_system_score_codex":0.0005731808,"about_ca_system_score_gemma":0.0006647637,"threshold_uncertainty_score":0.014326394},"labels":[],"label_agreement":null},{"id":"W1496855202","doi":"10.48550/arxiv.1301.6690","title":"Model-Based Bayesian Exploration","year":2013,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":235,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Reinforcement learning; Value of information; Action (physics); Bayesian probability; Quality (philosophy); Artificial intelligence; Value (mathematics); Machine learning; Probability distribution; Mathematical optimization; Mathematics; Statistics","score_opus":0.07631130262408777,"score_gpt":0.1767027802933835,"score_spread":0.10039147766929572,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1496855202","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010349255,0.0006014179,0.98062485,0.0007207014,0.000031704665,0.000039636918,0.00010767282,0.00026361446,0.007261127],"genre_scores_gemma":[0.77760696,0.0013332815,0.21224158,0.00036623253,0.00011772987,0.00031657098,0.00035461033,0.0001441359,0.007518826],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99876344,0.00058429106,0.000049972725,0.00021994692,0.0002925618,0.00008981128],"domain_scores_gemma":[0.9970394,0.002021674,0.00027261837,0.00025142578,0.00029839305,0.00011642048],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021332847,0.00096225744,0.0015770654,0.0010226391,0.0005901604,0.001980171,0.0017548943,0.0013855221,0.004955596],"category_scores_gemma":[0.009571326,0.00074411405,0.0008366693,0.000993052,0.0015467132,0.0030380806,0.0019152398,0.002010569,0.0007883726],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007831826,0.00004702756,0.00087988976,0.00012204538,0.00006733849,0.00006175652,0.000100811136,0.7935006,0.00046322672,0.16004702,0.0018126314,0.0428193],"study_design_scores_gemma":[0.000018741175,0.000018854582,0.000118256234,0.00002500906,0.00001133218,0.000023120732,0.000011728443,0.84002876,0.00019550146,0.15799136,0.0015463316,0.000011075732],"about_ca_topic_score_codex":0.00415092,"about_ca_topic_score_gemma":0.0055938014,"teacher_disagreement_score":0.004955596,"about_ca_system_score_codex":0.0018872782,"about_ca_system_score_gemma":0.0017743236,"threshold_uncertainty_score":0.016578078},"labels":[],"label_agreement":null},{"id":"W1498235675","doi":"","title":"AEMS: an anytime online search algorithm for approximate policy refinement in large POMDPs","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":65,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Partially observable Markov decision process; Markov decision process; Computation; Task (project management); Mathematical optimization; State space; State (computer science); Function (biology); Bellman equation; Online algorithm; Observable; Algorithm; Markov process; Markov chain; Machine learning; Markov model; Mathematics","score_opus":0.03203246700207627,"score_gpt":0.34282257742561884,"score_spread":0.31079011042354254,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1498235675","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0052230083,0.0001031477,0.9927654,0.000098107186,0.000028088887,0.000050567614,0.000036410715,0.00075086684,0.0009443356],"genre_scores_gemma":[0.36330363,0.00017564185,0.6330306,0.0001831661,0.000049896065,0.00046950276,0.00019861777,0.0002269723,0.0023620063],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99911505,0.00028177953,0.000064468346,0.00016911249,0.0002727166,0.000096809825],"domain_scores_gemma":[0.9978624,0.0014670191,0.00018865977,0.00020078382,0.00018057786,0.00010051765],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016557943,0.001084854,0.0015743446,0.0006587129,0.0004781042,0.00083453703,0.0021187821,0.0017012446,0.004150755],"category_scores_gemma":[0.005930704,0.000595987,0.00084196776,0.00066544133,0.000900804,0.0018986886,0.0022891904,0.0021127528,0.0007250078],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026517783,0.00011584178,0.00046833197,0.00012536978,0.00005539459,0.00008308498,0.00011617068,0.85835624,0.0018796853,0.02582916,0.0021163346,0.1105892],"study_design_scores_gemma":[0.000030710733,0.00002115738,0.00002519236,0.0000054890343,0.000003827522,0.0000095447585,0.0000057964753,0.9941298,0.00033062304,0.005026673,0.00040798646,0.000003296366],"about_ca_topic_score_codex":0.004158682,"about_ca_topic_score_gemma":0.004425732,"teacher_disagreement_score":0.004158682,"about_ca_system_score_codex":0.000920685,"about_ca_system_score_gemma":0.0021073911,"threshold_uncertainty_score":0.013885617},"labels":[],"label_agreement":null},{"id":"W1503109067","doi":"","title":"Point-Based Value Iteration for Constrained POMDPs","year":2012,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Markov decision process; Mathematical optimization; Bellman equation; Computer science; Dynamic programming; Minimax; Linear programming; Value (mathematics); Scalability; Function (biology); Point (geometry); Markov process; Mathematics","score_opus":0.02092790281422207,"score_gpt":0.26621454782228016,"score_spread":0.24528664500805808,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1503109067","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004368687,0.00008759745,0.9938578,0.00006882051,0.000015383472,0.000041540454,0.000029556317,0.00011358342,0.0014170934],"genre_scores_gemma":[0.5050951,0.00033967377,0.49020573,0.00012633995,0.000044149678,0.0006564924,0.00021327655,0.00017370823,0.0031455385],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984376,0.0007137562,0.000079372294,0.00023343303,0.00039189993,0.00014384383],"domain_scores_gemma":[0.9960419,0.0030871402,0.00022842844,0.00016846712,0.00035715406,0.00011688432],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025376242,0.0010964428,0.0017983065,0.0006454653,0.0005842845,0.0011976737,0.001465477,0.0013498623,0.003936662],"category_scores_gemma":[0.0078091766,0.0008369017,0.0010263035,0.00074172573,0.0019056294,0.0017386192,0.0019804083,0.0024117623,0.00041426654],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003460664,0.000019822553,0.00018276942,0.0000633154,0.000019782214,0.000032489253,0.000044109223,0.94183457,0.00033094556,0.045316774,0.00035610722,0.0117647145],"study_design_scores_gemma":[0.000008368846,0.000011129018,0.000013361739,0.000005495384,0.0000019767995,0.000003986475,0.0000039541883,0.9808853,0.00013201417,0.018674508,0.00025655248,0.0000033256033],"about_ca_topic_score_codex":0.005064134,"about_ca_topic_score_gemma":0.004529928,"teacher_disagreement_score":0.005064134,"about_ca_system_score_codex":0.0015681515,"about_ca_system_score_gemma":0.0019162801,"threshold_uncertainty_score":0.013420403},"labels":[],"label_agreement":null},{"id":"W150352456","doi":"10.5072/zenodo.49098","title":"Learning by Automatic Option Discovery from Conditionally Terminating Sequences","year":2006,"lang":"en","type":"article","venue":"OpenMETU (Middle East Technical University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Normalization property; Computer science; Reinforcement learning; Tree (set theory); Tree structure; Artificial intelligence; Machine learning; Theoretical computer science; Data structure; Programming language; Mathematics","score_opus":0.012963262884764256,"score_gpt":0.19508612386393304,"score_spread":0.18212286097916877,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W150352456","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05176372,0.00013909063,0.9451369,0.0001623537,0.000021094682,0.000056466408,0.00014933405,0.0015786033,0.0009923815],"genre_scores_gemma":[0.6673023,0.00012391388,0.33044136,0.00010787859,0.00002720158,0.00015790094,0.0006189168,0.00014093012,0.0010796463],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985285,0.0005436341,0.00009584856,0.00033121172,0.00037748358,0.00012343132],"domain_scores_gemma":[0.99138665,0.006466138,0.0006326285,0.00067941454,0.0005575967,0.000277608],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002062082,0.00069235923,0.0009369775,0.0009362445,0.0004895714,0.0008371355,0.0019607889,0.000952708,0.002155606],"category_scores_gemma":[0.011912362,0.0006413033,0.00078034523,0.0006763435,0.0012533726,0.0033560838,0.001581281,0.002121403,0.00047383152],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012202037,0.00037748422,0.010628427,0.00039937673,0.00014882047,0.00068293285,0.0006596086,0.40870613,0.019376853,0.07966899,0.003820266,0.47431082],"study_design_scores_gemma":[0.000036814392,0.00007349419,0.00044184583,0.00002743971,0.00001561274,0.00008857487,0.000024039482,0.941896,0.0050386395,0.05136786,0.00096465164,0.000024993196],"about_ca_topic_score_codex":0.001529393,"about_ca_topic_score_gemma":0.0026953737,"teacher_disagreement_score":0.002155606,"about_ca_system_score_codex":0.00061047974,"about_ca_system_score_gemma":0.001609725,"threshold_uncertainty_score":0.010905445},"labels":[],"label_agreement":null},{"id":"W1504915502","doi":"10.1609/aaai.v24i1.7751","title":"Using Bisimulation for Policy Transfer in MDPs","year":2010,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Office of Naval Research; Natural Sciences and Engineering Research Council of Canada","keywords":"Bisimulation; Computer science; Pessimism; Task (project management); Metric (unit); Markov decision process; Transfer (computing); Action (physics); Quality (philosophy); Theoretical computer science; Artificial intelligence; Markov process; Mathematics; Economics","score_opus":0.13511302919804696,"score_gpt":0.35650515512498543,"score_spread":0.22139212592693847,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1504915502","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029525306,0.00015076745,0.9683262,0.00021839363,0.000023531713,0.0000957544,0.000033373715,0.00047365378,0.0011529662],"genre_scores_gemma":[0.70917064,0.0002102614,0.28872,0.00017556442,0.00002850704,0.0006063104,0.00015676123,0.00022041869,0.0007114673],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99570197,0.0023183436,0.00034140126,0.00063330383,0.00071108143,0.00029388364],"domain_scores_gemma":[0.982463,0.012710504,0.0016039856,0.001627289,0.00091084145,0.00068438414],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007882175,0.0022380573,0.001968806,0.0018686543,0.0010471073,0.0019568652,0.0022738413,0.0021740298,0.0026720439],"category_scores_gemma":[0.03515433,0.0010643506,0.001294177,0.0010713486,0.0027746402,0.005382282,0.0049357507,0.003635802,0.00041400443],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016198438,0.00008569885,0.00055977475,0.000090176116,0.000053860997,0.00003425929,0.00014491413,0.90996313,0.0013816479,0.059368514,0.00020390884,0.027952034],"study_design_scores_gemma":[0.000026101245,0.00008366796,0.000055125805,0.000015537355,0.00000860264,0.000009179269,0.000014492672,0.9539768,0.0014092617,0.04409033,0.0002992935,0.000011650144],"about_ca_topic_score_codex":0.0024259456,"about_ca_topic_score_gemma":0.0015877155,"teacher_disagreement_score":0.007882175,"about_ca_system_score_codex":0.003690415,"about_ca_system_score_gemma":0.0027235004,"threshold_uncertainty_score":0.041685462},"labels":[],"label_agreement":null},{"id":"W1507852408","doi":"10.48550/arxiv.1206.3281","title":"Model-Based Bayesian Reinforcement Learning in Large Structured Domains","year":2012,"lang":"en","type":"preprint","venue":"PubMed","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"National Institute of Mental Health","keywords":"Reinforcement learning; Computer science; Scalability; Artificial intelligence; Bayesian probability; Machine learning; Bayesian inference","score_opus":0.025380733767491163,"score_gpt":0.24242802852349365,"score_spread":0.21704729475600248,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1507852408","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01856552,0.00027141746,0.97902644,0.00036744625,0.000019231638,0.000024885263,0.00004138932,0.00020365271,0.0014799437],"genre_scores_gemma":[0.8434151,0.0004961934,0.15333696,0.0001520397,0.000059968133,0.00016784332,0.00013675765,0.0000752232,0.0021600092],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991191,0.00044774433,0.00003440956,0.00014677597,0.00017228458,0.000079642574],"domain_scores_gemma":[0.99587846,0.0031352097,0.00031346935,0.00024952853,0.0002384954,0.00018485471],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002022441,0.00075568206,0.0014484319,0.00050908444,0.00044781395,0.0010000804,0.0012764181,0.0013215421,0.0019813161],"category_scores_gemma":[0.010272616,0.0007143003,0.00061148166,0.0005699986,0.0018942028,0.0023671135,0.0017074122,0.0021432512,0.0002295194],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000043753218,0.000025473293,0.00027439703,0.00004572865,0.000021029493,0.000050191593,0.000051507897,0.9386036,0.00037065582,0.04780871,0.00044456095,0.012260384],"study_design_scores_gemma":[0.000011038513,0.00000918371,0.000034661203,0.0000043785317,0.0000023100586,0.0000055459523,0.00000389704,0.96135455,0.000080376725,0.038334586,0.00015614563,0.0000032509358],"about_ca_topic_score_codex":0.006967803,"about_ca_topic_score_gemma":0.006428736,"teacher_disagreement_score":0.006967803,"about_ca_system_score_codex":0.0013885874,"about_ca_system_score_gemma":0.0012127756,"threshold_uncertainty_score":0.013854504},"labels":[],"label_agreement":null},{"id":"W1508997545","doi":"10.1609/aimag.v31i2.2227","title":"The Reinforcement Learning Competitions","year":2010,"lang":"en","type":"article","venue":"AI Magazine","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"National Institute of Standards and Technology; University of Southern California; Toyota Research Institute, North America; Toyota Research Institute; University of Alberta; International Business Machines Corporation","keywords":"Reinforcement learning; Competition (biology); Computer science; Reinforcement; Focus (optics); Data science; Artificial intelligence; Engineering","score_opus":0.007611555982384794,"score_gpt":0.24271572478864836,"score_spread":0.23510416880626356,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1508997545","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11677143,0.011690483,0.48817104,0.047616605,0.005147844,0.0019303412,0.0019077528,0.0020074812,0.32475713],"genre_scores_gemma":[0.78835374,0.0025534646,0.13904524,0.012668365,0.0017547077,0.002420018,0.001990829,0.000828713,0.050385013],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.96763057,0.01704049,0.001269853,0.0028503218,0.009356865,0.0018518943],"domain_scores_gemma":[0.9350808,0.043310143,0.002430627,0.00451651,0.010136512,0.0045254305],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.028121466,0.0013060773,0.0017332432,0.0011554582,0.002612522,0.0054923315,0.00311798,0.0037080154,0.013016673],"category_scores_gemma":[0.08247094,0.00051637884,0.001347997,0.0010750316,0.0041657616,0.0069193784,0.0050234157,0.006970925,0.0019211832],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012607884,0.0009614907,0.0039933054,0.0006466505,0.0002838637,0.00021795648,0.0006971844,0.052906625,0.0014816581,0.61140186,0.10648725,0.21966141],"study_design_scores_gemma":[0.0008378826,0.0013715742,0.004295197,0.0005161904,0.00013577154,0.00034255773,0.00057859375,0.19469836,0.0038162905,0.53135407,0.2617937,0.00025981406],"about_ca_topic_score_codex":0.0068945466,"about_ca_topic_score_gemma":0.0065401294,"teacher_disagreement_score":0.028121466,"about_ca_system_score_codex":0.007139361,"about_ca_system_score_gemma":0.0056459815,"threshold_uncertainty_score":0.14872235},"labels":[],"label_agreement":null},{"id":"W1511842811","doi":"10.1007/11766247_31","title":"Partial Local FriendQ Multiagent Learning: Application to Team Automobile Coordination Problem","year":2006,"lang":"en","type":"article","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Observability; Reinforcement learning; Computer science; Multi-agent system; Degree (music); Observable; Mathematical optimization; Computation; Artificial intelligence; Mathematics; Algorithm; Applied mathematics","score_opus":0.006367876696141645,"score_gpt":0.24259299922135455,"score_spread":0.2362251225252129,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1511842811","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08679021,0.0002445054,0.90871096,0.00037571415,0.000044757515,0.00006437373,0.00005878963,0.0005398497,0.003170787],"genre_scores_gemma":[0.88147205,0.00010438628,0.11544194,0.00008656733,0.000041392155,0.00015046453,0.00010698674,0.00008983689,0.0025063807],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99960107,0.00018368404,0.000017223843,0.00007488321,0.000075221906,0.00004788505],"domain_scores_gemma":[0.9973908,0.0017807061,0.00016287676,0.00018266871,0.00031410952,0.00016887474],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021253787,0.00066359836,0.0012445474,0.00048958964,0.0009265093,0.00063299085,0.00157574,0.001314314,0.0030854738],"category_scores_gemma":[0.006171744,0.00037835896,0.00043356928,0.00068242155,0.00077442365,0.0012661266,0.0019119785,0.00088449806,0.00025199275],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012102444,0.00009524153,0.0006864483,0.00008525859,0.000032433916,0.00006649305,0.00010213785,0.9299537,0.000531355,0.012039475,0.0017510969,0.054535467],"study_design_scores_gemma":[0.000012554848,0.000024099249,0.000044823257,0.0000014428542,0.0000025539728,0.000005210005,0.00000766372,0.9957496,0.0000825869,0.0039587254,0.00010912445,0.0000015766353],"about_ca_topic_score_codex":0.0050243908,"about_ca_topic_score_gemma":0.0039124126,"teacher_disagreement_score":0.0050243908,"about_ca_system_score_codex":0.0006982553,"about_ca_system_score_gemma":0.00078466185,"threshold_uncertainty_score":0.011240244},"labels":[],"label_agreement":null},{"id":"W1512866498","doi":"10.1609/aaai.v26i1.8321","title":"Investigating Contingency Awareness Using Atari 2600 Games","year":2021,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":76,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"University of Alberta; Alberta Innovates; Western Canada Research Grid; Compute Canada","keywords":"Contingency; Reinforcement learning; Exploit; Computer science; Function (biology); Control (management); Value (mathematics); Reinforcement; Artificial intelligence; Contingency management; Bellman equation; Human–computer interaction; Machine learning; Psychology; Social psychology; Computer security","score_opus":0.13878597859183864,"score_gpt":0.32726220209429063,"score_spread":0.188476223502452,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1512866498","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.93271923,0.00012339704,0.06183095,0.00022078754,0.000024216284,0.00014562024,0.00018040839,0.00022374639,0.0045317183],"genre_scores_gemma":[0.978865,0.000028794248,0.02011225,0.00003834449,0.000004557876,0.00005627756,0.00013404885,0.000017610855,0.0007431327],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989699,0.00057016045,0.000032243865,0.00016341398,0.00017907766,0.000085165404],"domain_scores_gemma":[0.9949379,0.003808554,0.00036190596,0.00032478248,0.00027364065,0.00029328951],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018589629,0.0007595566,0.0006256036,0.00047799308,0.0004931751,0.0009255846,0.0012602984,0.0006194917,0.0016279907],"category_scores_gemma":[0.011444485,0.00029352307,0.0004782608,0.00028682838,0.0011109541,0.0015827655,0.0012760662,0.0016778759,0.00012900293],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010811877,0.0011010434,0.017340956,0.00024817305,0.00025209322,0.00041105863,0.001276351,0.879957,0.0077755973,0.030537589,0.0022133328,0.057805568],"study_design_scores_gemma":[0.000046108067,0.00028210293,0.0021418536,0.0000096304575,0.000012138991,0.00003159877,0.00014539734,0.9865964,0.0013048786,0.008717217,0.000695378,0.000017272183],"about_ca_topic_score_codex":0.008297101,"about_ca_topic_score_gemma":0.009641239,"teacher_disagreement_score":0.008297101,"about_ca_system_score_codex":0.0011875355,"about_ca_system_score_gemma":0.0006866451,"threshold_uncertainty_score":0.016497612},"labels":[],"label_agreement":null},{"id":"W1513366144","doi":"10.1007/11527503_11","title":"Multiagent Association Rules Mining in Cooperative Learning Systems","year":2005,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Reinforcement learning; Artificial intelligence; Robustness (evolution); Association rule learning; Machine learning; Multi-agent system; Domain (mathematical analysis)","score_opus":0.017857517415009107,"score_gpt":0.247376344211483,"score_spread":0.2295188267964739,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1513366144","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022890382,0.0009325631,0.97424376,0.00023639703,0.000062671184,0.00009295694,0.00006846153,0.00043948521,0.0010332934],"genre_scores_gemma":[0.37529945,0.0006174406,0.61947644,0.00014379101,0.00010560615,0.00031633553,0.0004208891,0.00009845359,0.0035216608],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9972283,0.0011677678,0.00030057988,0.00049894373,0.000685148,0.00011934576],"domain_scores_gemma":[0.9896578,0.007767667,0.00048579325,0.0008995163,0.0010086809,0.00018047901],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041400483,0.0007368553,0.0016033403,0.0016591003,0.00084565795,0.0015597073,0.0029584104,0.001348937,0.0014577764],"category_scores_gemma":[0.012455898,0.00083548785,0.0010272054,0.001709631,0.0008090606,0.0023504035,0.0019813192,0.0016836791,0.0006777052],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025706133,0.0005262106,0.0070702597,0.00029008262,0.000368789,0.0003455645,0.00039862102,0.31964007,0.0027432733,0.013976224,0.0040926877,0.6502912],"study_design_scores_gemma":[0.000014622884,0.000043413987,0.00063253846,0.000022693705,0.00003788127,0.000118561264,0.000055842083,0.97655225,0.001734388,0.01957953,0.0011940458,0.000014163851],"about_ca_topic_score_codex":0.0022978976,"about_ca_topic_score_gemma":0.0023209034,"teacher_disagreement_score":0.0041400483,"about_ca_system_score_codex":0.00056618697,"about_ca_system_score_gemma":0.00081593724,"threshold_uncertainty_score":0.021894932},"labels":[],"label_agreement":null},{"id":"W1517412219","doi":"10.1109/nafips.2001.944350","title":"On playing games without knowing the rules","year":2002,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Premise; Salient; Adversary; Task (project management); Metagaming; Video game design; Non-cooperative game; Artificial intelligence; Screening game; Game mechanics; Game design; Repeated game; Game theory; Mechanism (biology); Sequential game; Human–computer interaction; Combinatorial game theory; Simultaneous game; Computer security; Mathematical economics; Engineering; Mathematics; Epistemology","score_opus":0.029908834774834876,"score_gpt":0.24536985819494986,"score_spread":0.21546102342011497,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1517412219","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12024417,0.003579898,0.56630206,0.019526465,0.000542235,0.00036355367,0.00025101582,0.0009846794,0.28820592],"genre_scores_gemma":[0.83701634,0.0035605396,0.119732484,0.0023754197,0.00036749314,0.00050474395,0.00026992508,0.0001896248,0.035983495],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984242,0.0007343549,0.00006410409,0.0002865345,0.00034064346,0.00015012553],"domain_scores_gemma":[0.9968028,0.001963505,0.00021928581,0.00044248227,0.0002945679,0.00027744626],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001396026,0.0009772531,0.00059571216,0.00042993738,0.00081079704,0.0029856607,0.0012046559,0.0016557304,0.006557965],"category_scores_gemma":[0.009617879,0.0004085581,0.00039730742,0.00028060717,0.0050589712,0.007034947,0.0026739952,0.0026292421,0.0014448625],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003006991,0.00022255965,0.0021567224,0.00036407707,0.00006984322,0.00019974183,0.002489964,0.031386245,0.0039161113,0.822639,0.009994157,0.12626094],"study_design_scores_gemma":[0.000090172405,0.00015290777,0.0012477011,0.00016511307,0.00003253077,0.00016887458,0.00060758047,0.07842872,0.0014922364,0.8785715,0.038983654,0.00005896649],"about_ca_topic_score_codex":0.0019215259,"about_ca_topic_score_gemma":0.0016125338,"teacher_disagreement_score":0.006557965,"about_ca_system_score_codex":0.00086578634,"about_ca_system_score_gemma":0.0008019521,"threshold_uncertainty_score":0.021938622},"labels":[],"label_agreement":null},{"id":"W1519610715","doi":"10.1109/ijcnn.1999.833418","title":"Reinforcement learning for autonomous robot navigation","year":2003,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Bellman equation; Computer science; Mobile robot; Motion planning; Robot; Piecewise linear function; Robot learning; Artificial intelligence; Function (biology); Function approximation; Piecewise; Path (computing); Mobile robot navigation; Robot control; Mathematical optimization; Mathematics; Artificial neural network","score_opus":0.02127075616356644,"score_gpt":0.26146204335521966,"score_spread":0.24019128719165322,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1519610715","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008129226,0.0037521813,0.9811109,0.0007979075,0.00013580876,0.000027083173,0.0000310339,0.0003501714,0.005665743],"genre_scores_gemma":[0.83349574,0.004045166,0.15182047,0.00027867864,0.00033188314,0.00032226183,0.00012696328,0.000097595796,0.009481292],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99966943,0.00014591086,0.000017948842,0.000049726263,0.00008753622,0.000029371477],"domain_scores_gemma":[0.9990544,0.0006940487,0.00006481679,0.000046677018,0.00009448711,0.00004570358],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007808614,0.0006015618,0.00081501686,0.0002628293,0.00026942726,0.00073662365,0.0006511485,0.0007844234,0.0021831994],"category_scores_gemma":[0.0029062005,0.00022033077,0.00038668848,0.00036456538,0.00090168376,0.0007251042,0.0006216538,0.0018929681,0.00041289712],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009447141,0.000060915452,0.00046321115,0.00018605834,0.000062452455,0.0001078918,0.00011396579,0.7200537,0.001610124,0.1601435,0.00362294,0.1134808],"study_design_scores_gemma":[0.000025502613,0.000036696278,0.00008287856,0.000015939368,0.000007982376,0.000018527253,0.000008433458,0.8907912,0.00030223993,0.10596712,0.0027356097,0.0000078754265],"about_ca_topic_score_codex":0.0033095824,"about_ca_topic_score_gemma":0.0016632638,"teacher_disagreement_score":0.0033095824,"about_ca_system_score_codex":0.0011187753,"about_ca_system_score_gemma":0.0006144178,"threshold_uncertainty_score":0.008117378},"labels":[],"label_agreement":null},{"id":"W1521332421","doi":"","title":"A Bayesian approach to imitation in reinforcement learning","year":2003,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; University of British Columbia","funders":"","keywords":"Imitation; Computer science; Bayesian probability; Reinforcement learning; Artificial intelligence; Machine learning; Bayesian network; Bayesian inference; Social learning; Knowledge management; Psychology","score_opus":0.019689779485534634,"score_gpt":0.24364885994542235,"score_spread":0.22395908045988772,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1521332421","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001698911,0.0009070921,0.99080503,0.0010765906,0.00008914612,0.000022617443,0.000032838776,0.00006561297,0.005302145],"genre_scores_gemma":[0.5481937,0.0047545107,0.42964154,0.00096458703,0.001082676,0.00060996023,0.00014983484,0.00017210183,0.014431206],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9970708,0.0016159308,0.0001423709,0.0003629728,0.00065479754,0.0001530571],"domain_scores_gemma":[0.9938937,0.004614142,0.00041184112,0.00036209924,0.00044479623,0.0002733795],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038239826,0.0012828857,0.0016026604,0.001055199,0.0007851289,0.0019125282,0.0027461122,0.0027545164,0.0043368605],"category_scores_gemma":[0.014405796,0.00086861436,0.001181233,0.0011817709,0.0040039048,0.004970582,0.0024317433,0.003933884,0.0007880796],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000037685688,0.000037016005,0.0003039285,0.00010012934,0.000056670226,0.00010604813,0.00019757189,0.15010765,0.000439988,0.8260015,0.0014102878,0.021201538],"study_design_scores_gemma":[0.000025667865,0.00003158153,0.000086477245,0.000027358863,0.00001875256,0.000041381547,0.000016905857,0.33733037,0.00017587871,0.6589066,0.0033159654,0.000023085915],"about_ca_topic_score_codex":0.0036144624,"about_ca_topic_score_gemma":0.0033153,"teacher_disagreement_score":0.0043368605,"about_ca_system_score_codex":0.0017096789,"about_ca_system_score_gemma":0.0013601849,"threshold_uncertainty_score":0.02022338},"labels":[],"label_agreement":null},{"id":"W1526654727","doi":"","title":"Model-based reinforcement learning with nearly tight exploration complexity bounds","year":2010,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":105,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Athabasca University; University of Alberta","funders":"","keywords":"Reinforcement learning; Markov decision process; Upper and lower bounds; Logarithm; Computer science; Probably approximately correct learning; Markov process; Sample complexity; Contrast (vision); Binary logarithm; Exploratory analysis; Q-learning; Algorithm; Exploratory research; Markov chain; Process (computing); Mathematics; Artificial intelligence; Combinatorics; Machine learning; Active learning (machine learning); Computational learning theory; Statistics","score_opus":0.04441804103927028,"score_gpt":0.2579625849142766,"score_spread":0.21354454387500632,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1526654727","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024770245,0.0010155404,0.96708995,0.000818133,0.000057347424,0.00006538086,0.00007087171,0.0008329722,0.005279442],"genre_scores_gemma":[0.8122701,0.0008706289,0.18012896,0.0005417372,0.00017247605,0.00041391232,0.0003030687,0.00038844268,0.0049107363],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9966839,0.0010797265,0.00015630526,0.00057424017,0.0010295445,0.00047629222],"domain_scores_gemma":[0.9811255,0.015115006,0.0009648556,0.0015670198,0.0006939395,0.0005336257],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043917317,0.0020561775,0.0028031,0.00084879017,0.0007190059,0.0024078325,0.0022346345,0.0022021604,0.0033123535],"category_scores_gemma":[0.023349203,0.0010489853,0.0013988054,0.00082589756,0.002198512,0.0048839315,0.0044460893,0.0055474755,0.0007712414],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020105347,0.00012712165,0.00078374794,0.00015592598,0.00006851704,0.00006392212,0.000095548996,0.9112448,0.0011478773,0.055481598,0.0015528669,0.02907709],"study_design_scores_gemma":[0.00001692431,0.000031690164,0.000043517834,0.000007728079,0.0000066938696,0.000011982959,0.0000033562153,0.9734664,0.00018798118,0.026048861,0.00016968488,0.0000051766833],"about_ca_topic_score_codex":0.0029216853,"about_ca_topic_score_gemma":0.0029736483,"teacher_disagreement_score":0.0043917317,"about_ca_system_score_codex":0.0023562168,"about_ca_system_score_gemma":0.0023975007,"threshold_uncertainty_score":0.023225963},"labels":[],"label_agreement":null},{"id":"W1530545262","doi":"10.1007/978-3-642-04380-2_84","title":"A Real-Time Transfer and Adaptive Learning Approach for Game Agents in a Layered Architecture","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Architecture; Transfer of learning; Artificial intelligence; Human–computer interaction; Real-time computing; Distributed computing; Computer architecture","score_opus":0.026707445206418993,"score_gpt":0.249829523060819,"score_spread":0.2231220778544,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1530545262","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008962427,0.00004477678,0.98850983,0.00009555943,0.000024518156,0.00003115661,0.000009488611,0.0003511158,0.0019711126],"genre_scores_gemma":[0.5892192,0.00011080843,0.3998875,0.00012229774,0.000041779258,0.00025465252,0.000048572692,0.00013148002,0.010183692],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99964607,0.00010162454,0.000020709953,0.00006334303,0.00010620908,0.00006206196],"domain_scores_gemma":[0.9993987,0.0003045644,0.000037709153,0.00007130795,0.00012964655,0.000058016645],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00088292424,0.0006108746,0.0007249084,0.00028729034,0.00049712224,0.0009083166,0.0025908833,0.0013240158,0.004344675],"category_scores_gemma":[0.002339152,0.0004196905,0.0006394693,0.00030402085,0.00096795516,0.0015494381,0.0022594247,0.0019783257,0.0005848731],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000109963425,0.0001158948,0.00029810224,0.000065615466,0.000062633524,0.00009889364,0.00018643717,0.83212584,0.0064112456,0.05724783,0.0015399224,0.101737656],"study_design_scores_gemma":[0.000004280845,0.000013056443,0.00001989934,0.0000014800161,0.0000039241677,0.000006745436,0.0000043816685,0.993383,0.00031680023,0.0060708583,0.00017244478,0.000003118161],"about_ca_topic_score_codex":0.0053620166,"about_ca_topic_score_gemma":0.0047263633,"teacher_disagreement_score":0.0053620166,"about_ca_system_score_codex":0.00094441115,"about_ca_system_score_gemma":0.0009337238,"threshold_uncertainty_score":0.014534354},"labels":[],"label_agreement":null},{"id":"W1530573725","doi":"10.48550/arxiv.1206.6879","title":"Practical Linear Value-approximation Techniques for First-order MDPs","year":2012,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Markov decision process; Computer science; Bellman equation; Mathematical optimization; Linear programming; Probabilistic logic; Set (abstract data type); Function (biology); Domain (mathematical analysis); Value (mathematics); Markov process; Order (exchange); Algorithm; Mathematics; Artificial intelligence; Machine learning","score_opus":0.09509342492505367,"score_gpt":0.23970681743159786,"score_spread":0.1446133925065442,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1530573725","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0043244297,0.00023596492,0.9931385,0.00019033307,0.000014351325,0.00003648844,0.000027936558,0.00023206529,0.001799887],"genre_scores_gemma":[0.34505266,0.0006002193,0.6498432,0.00023501004,0.000069179,0.0004517939,0.0001849443,0.00030911702,0.003253884],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99814796,0.00088257727,0.00007974262,0.00020664852,0.00048583455,0.00019724482],"domain_scores_gemma":[0.9926267,0.0060542645,0.00034415847,0.00044502047,0.00040205938,0.00012783874],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003967328,0.0014605875,0.0017701498,0.0009826598,0.0006775125,0.001517744,0.0019080498,0.0015304222,0.0044678543],"category_scores_gemma":[0.012386158,0.0009678878,0.0012650165,0.0013382497,0.001588983,0.0023714225,0.0021250248,0.004364605,0.0007516555],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000047614576,0.00006166828,0.00033004355,0.000114300434,0.000030567953,0.000029884774,0.000089146655,0.8942477,0.00041964874,0.06275066,0.001113792,0.0407649],"study_design_scores_gemma":[0.000009218928,0.000012544212,0.00001859164,0.000011508955,0.000003880949,0.0000069497723,0.00000918353,0.97265834,0.00020521441,0.026689189,0.00037247787,0.000002816007],"about_ca_topic_score_codex":0.005434577,"about_ca_topic_score_gemma":0.007759475,"teacher_disagreement_score":0.005434577,"about_ca_system_score_codex":0.0024265621,"about_ca_system_score_gemma":0.0022647327,"threshold_uncertainty_score":0.02098149},"labels":[],"label_agreement":null},{"id":"W1535857328","doi":"10.1002/9780470724163.ch30","title":"Approximation and Perception in Ethology‐Based Reinforcement Learning","year":2008,"lang":"en","type":"other","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Ethology; Reinforcement; Perception; Reinforcement learning; Animal learning; Psychology; Cognitive psychology; Artificial intelligence; Computer science; Cognitive science; Neuroscience; Social psychology; Biology; Ecology","score_opus":0.018992809401622634,"score_gpt":0.2527448291641516,"score_spread":0.23375201976252896,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1535857328","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005714888,0.005423058,0.97153574,0.0014167982,0.00038817676,0.000020595702,0.000054142984,0.00011452073,0.01533205],"genre_scores_gemma":[0.57850856,0.011082994,0.37893954,0.00044044465,0.00082985696,0.00019173333,0.00022500326,0.00021820587,0.029563699],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996706,0.00014389544,0.000019254447,0.000059688606,0.00008477841,0.000021779291],"domain_scores_gemma":[0.9994031,0.0003822264,0.00003642605,0.00006135665,0.000084551866,0.000032252396],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012505339,0.0005130287,0.0007735603,0.00047699394,0.00028477277,0.0018593824,0.0009943561,0.0009444409,0.004209523],"category_scores_gemma":[0.002689961,0.00037790387,0.0007649132,0.0007464301,0.0020765746,0.0029489922,0.0008970004,0.0020784242,0.0005984306],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003059864,0.000025691847,0.0002788171,0.00014274727,0.000042546755,0.000035504487,0.00012421035,0.13203964,0.00087986956,0.80177593,0.0022986503,0.06232586],"study_design_scores_gemma":[0.000009231351,0.000031771517,0.00032593348,0.00004640684,0.000012889049,0.000029831923,0.000045304772,0.33335522,0.0006733195,0.65717334,0.008274926,0.000021687729],"about_ca_topic_score_codex":0.002518338,"about_ca_topic_score_gemma":0.0012293396,"teacher_disagreement_score":0.004209523,"about_ca_system_score_codex":0.0010854955,"about_ca_system_score_gemma":0.0006340173,"threshold_uncertainty_score":0.014082253},"labels":[],"label_agreement":null},{"id":"W1537507437","doi":"10.1007/11766247_42","title":"The K Best-Paths Approach to Approximate Dynamic Programming with Application to Portfolio Optimization","year":2006,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Dynamic programming; Reinforcement learning; Mathematical optimization; Markov decision process; Portfolio; Kernel (algebra); Controller (irrigation); Sharpe ratio; Project portfolio management; Portfolio optimization; Artificial intelligence; Markov process; Algorithm; Mathematics; Finance; Project management","score_opus":0.008187980339184919,"score_gpt":0.22828879796239565,"score_spread":0.22010081762321074,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1537507437","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0020817963,0.0005582009,0.99507326,0.00015681727,0.000045270444,0.000014043804,0.000022063685,0.00008349688,0.001965076],"genre_scores_gemma":[0.21683522,0.0020748961,0.77045745,0.00017612881,0.00019464974,0.0003190157,0.00016166698,0.00027751317,0.009503477],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99905676,0.0005241621,0.00004546697,0.00013475351,0.00018616267,0.000052738338],"domain_scores_gemma":[0.9970475,0.0022730879,0.00013945853,0.0001643621,0.0002662868,0.00010940047],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023648187,0.0015691125,0.0020843223,0.0011789235,0.0008165988,0.0016940362,0.0021793947,0.0026148702,0.004389616],"category_scores_gemma":[0.0109311035,0.0013887997,0.0012400569,0.002405591,0.0020168985,0.002937543,0.0025803274,0.0033408422,0.0006769713],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000046067762,0.00004633194,0.00034318745,0.00009138483,0.000059343605,0.00004672459,0.00006153781,0.8111172,0.0002536834,0.12192557,0.0019469026,0.0640621],"study_design_scores_gemma":[0.000006384874,0.00000924496,0.000028250077,0.0000071901213,0.0000051747625,0.000010733089,0.000004504538,0.9400777,0.000046878777,0.059251346,0.00054724613,0.0000052519035],"about_ca_topic_score_codex":0.0055628116,"about_ca_topic_score_gemma":0.005585606,"teacher_disagreement_score":0.0055628116,"about_ca_system_score_codex":0.0012910868,"about_ca_system_score_gemma":0.0016168661,"threshold_uncertainty_score":0.014684737},"labels":[],"label_agreement":null},{"id":"W1537974983","doi":"","title":"Reinforcement using supervised learning for policy generalization","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Reinforcement learning; Markov decision process; Bellman equation; Computer science; Function approximation; Q-learning; Artificial intelligence; Mathematical optimization; Temporal difference learning; Generalization; Machine learning; Formalism (music); Markov process; Mathematics; Artificial neural network","score_opus":0.03822809926627745,"score_gpt":0.3112965769924141,"score_spread":0.2730684777261366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1537974983","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010814878,0.00023811843,0.98699677,0.00019177124,0.00003592552,0.00004537749,0.000032876564,0.0006053786,0.0010389349],"genre_scores_gemma":[0.8439763,0.00032601185,0.15232438,0.00022693409,0.000111291236,0.00033905107,0.00020379871,0.00015359346,0.0023385712],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99904317,0.00040531027,0.000057706802,0.00021018338,0.00020476716,0.00007882967],"domain_scores_gemma":[0.99391335,0.004354985,0.00040441818,0.00069786824,0.00047795792,0.00015142767],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002275392,0.0008380432,0.0018523919,0.00060599425,0.00045848265,0.00061887706,0.0015415399,0.0013041364,0.0027326297],"category_scores_gemma":[0.00939098,0.00063200167,0.0006538919,0.00056924176,0.0012643611,0.001612104,0.0014258933,0.0024404875,0.000505085],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010296562,0.00011335284,0.0006252574,0.000102346305,0.00006680322,0.000050871833,0.000061110855,0.90128267,0.00088625395,0.017478663,0.0015942154,0.077635504],"study_design_scores_gemma":[0.000007466238,0.00000943667,0.000025126492,0.000002856709,0.0000023759796,0.0000034224713,0.0000012850533,0.99425286,0.00015409556,0.005431781,0.000107110856,0.0000020637278],"about_ca_topic_score_codex":0.004357251,"about_ca_topic_score_gemma":0.0037763426,"teacher_disagreement_score":0.004357251,"about_ca_system_score_codex":0.0010577363,"about_ca_system_score_gemma":0.0017169922,"threshold_uncertainty_score":0.012033582},"labels":[],"label_agreement":null},{"id":"W1538703232","doi":"10.1007/978-3-642-04428-1_39","title":"Anytime Self-play Learning to Satisfy Functional Optimality Criteria","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Artificial intelligence; Mathematical optimization; Mathematics","score_opus":0.01797993096656769,"score_gpt":0.2577936049483812,"score_spread":0.23981367398181352,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1538703232","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043648023,0.00025289762,0.9266351,0.00040611762,0.00012850558,0.000109623834,0.00011597339,0.00030934287,0.028394401],"genre_scores_gemma":[0.8041073,0.0003069969,0.15384167,0.00025134545,0.00009149772,0.00036877205,0.00028720024,0.0003801651,0.040365],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99926406,0.00019113172,0.00004822061,0.00012768853,0.00022849263,0.00014042175],"domain_scores_gemma":[0.9984419,0.0008671642,0.00009000379,0.00012506125,0.00030535768,0.00017052978],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014413932,0.0011453398,0.0010107486,0.0003932073,0.00053119543,0.0011633951,0.0014831243,0.0012216519,0.007843879],"category_scores_gemma":[0.004461483,0.0003849897,0.0006230886,0.00035406672,0.0012015038,0.0020105077,0.0020898578,0.0019359741,0.00079532736],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022916571,0.0001626298,0.00045670348,0.0002096423,0.000045087676,0.0001227093,0.00030121123,0.121795274,0.0068025524,0.80592257,0.0051645976,0.058787815],"study_design_scores_gemma":[0.000038526035,0.00015462308,0.00016002152,0.000026779388,0.000011147936,0.00006349423,0.000046448673,0.63019955,0.0020469788,0.36494914,0.0022915283,0.000011729726],"about_ca_topic_score_codex":0.0014476811,"about_ca_topic_score_gemma":0.0015573151,"teacher_disagreement_score":0.007843879,"about_ca_system_score_codex":0.000978022,"about_ca_system_score_gemma":0.0012332716,"threshold_uncertainty_score":0.026240408},"labels":[],"label_agreement":null},{"id":"W1541317966","doi":"10.1609/aaai.v24i1.7740","title":"Robust Policy Computation in Reward-Uncertain MDPs Using Nondominated Policies","year":2010,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":47,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Regret; Markov decision process; Minimax; Computer science; Mathematical optimization; Set (abstract data type); Computation; Parallels; Exploit; Mathematics; Markov process; Algorithm; Machine learning; Economics","score_opus":0.12308326003183422,"score_gpt":0.3346686475142595,"score_spread":0.21158538748242528,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1541317966","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029321834,0.00010550763,0.9688779,0.00018890522,0.000020296544,0.000057951853,0.00007805253,0.00034377587,0.0010058031],"genre_scores_gemma":[0.5715406,0.000221616,0.42643264,0.0001175901,0.000026676817,0.0002756169,0.00026502597,0.00017008491,0.0009501011],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99809796,0.00073306344,0.00014855212,0.0004614026,0.00036416991,0.00019472005],"domain_scores_gemma":[0.98623544,0.010962852,0.0008195315,0.0012085261,0.00046480703,0.00030874115],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0047403607,0.0012870333,0.0018473363,0.00084165373,0.0006756426,0.002160882,0.0015345132,0.001471088,0.003252332],"category_scores_gemma":[0.02550971,0.0010835233,0.0013797343,0.0009105396,0.0020811774,0.0038843418,0.0029484464,0.003078257,0.0003093014],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010914462,0.00003081679,0.00036287252,0.000060130358,0.000022972665,0.000040550924,0.000054735734,0.95705956,0.0005735706,0.028576836,0.00020954676,0.012899287],"study_design_scores_gemma":[0.000015658296,0.00002456796,0.00004190389,0.000011571994,0.0000040810546,0.0000071089034,0.000009541075,0.97399205,0.0006280937,0.025123218,0.00013732055,0.00000494349],"about_ca_topic_score_codex":0.0030080718,"about_ca_topic_score_gemma":0.0032602113,"teacher_disagreement_score":0.0047403607,"about_ca_system_score_codex":0.0019259691,"about_ca_system_score_gemma":0.002689141,"threshold_uncertainty_score":0.025069714},"labels":[],"label_agreement":null},{"id":"W1544231892","doi":"10.48550/arxiv.1212.2471","title":"Monte Carlo Matrix Inversion Policy Evaluation","year":2012,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Monte Carlo method; Computer science; Inverse; Inversion (geology); Algorithm; Matrix (chemical analysis); Mathematical optimization; Mathematics; Statistics","score_opus":0.08529009758601776,"score_gpt":0.22615303763203295,"score_spread":0.14086294004601518,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1544231892","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009954851,0.00018333281,0.9829791,0.00030343095,0.00008430193,0.00018866059,0.00008389826,0.0014270609,0.004795372],"genre_scores_gemma":[0.34679708,0.00017968219,0.6468597,0.00032653648,0.00007889427,0.00066662533,0.0004506294,0.00056636246,0.0040745214],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978131,0.0008557514,0.00013319412,0.00027852505,0.0006445858,0.00027477683],"domain_scores_gemma":[0.9896952,0.0071223835,0.0003239995,0.0006798841,0.0019509152,0.00022763103],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036683136,0.0011156247,0.0017528438,0.0011502535,0.000921327,0.0014667057,0.0020892175,0.0015561114,0.010051705],"category_scores_gemma":[0.02404809,0.00068846403,0.0006776808,0.001033982,0.00092564186,0.0017092632,0.0014770918,0.002309715,0.0014648304],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023641378,0.00014493948,0.001522624,0.00013039383,0.00005398418,0.00007533702,0.00007657348,0.8241315,0.0013874212,0.03506957,0.003685174,0.13348602],"study_design_scores_gemma":[0.00002101214,0.000017138524,0.00005515194,0.0000075235484,0.0000047446715,0.000011120971,0.000008639655,0.99193734,0.0008133306,0.006496219,0.00062249403,0.0000053006306],"about_ca_topic_score_codex":0.009101712,"about_ca_topic_score_gemma":0.009452067,"teacher_disagreement_score":0.010051705,"about_ca_system_score_codex":0.0022507901,"about_ca_system_score_gemma":0.00419482,"threshold_uncertainty_score":0.03362626},"labels":[],"label_agreement":null},{"id":"W1545201153","doi":"","title":"Exploiting the Fascination: Video Games in Machine Learning Research and Education","year":2004,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Perspective (graphical); Turns, rounds and time-keeping systems in games; Point (geometry); Computer science; Video game; Game mechanics; Multimedia; Artificial neural network; Artificial intelligence; Video game design; Mathematics education; Psychology; Mathematics","score_opus":0.051549863138243085,"score_gpt":0.3323159438679224,"score_spread":0.28076608072967935,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1545201153","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08498834,0.027080918,0.7069882,0.04895821,0.002140894,0.00015729465,0.000092869275,0.0005263709,0.12906693],"genre_scores_gemma":[0.80173117,0.013458524,0.16156115,0.002949848,0.0014247607,0.00020794706,0.00012492675,0.00018889192,0.018352805],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9985929,0.0010019386,0.000038944378,0.000117720105,0.00017186276,0.000076641685],"domain_scores_gemma":[0.9960775,0.0030818656,0.00012983817,0.00023363403,0.00024728358,0.00022986626],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019820253,0.0007321719,0.00038561854,0.0007317607,0.0009413608,0.003086119,0.0007368796,0.0011209436,0.0028606823],"category_scores_gemma":[0.010426689,0.00029256256,0.0002806398,0.0005935896,0.0034357372,0.0055110874,0.002162652,0.002066289,0.00068580895],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029518353,0.00011918324,0.0024684945,0.00039412602,0.00006547971,0.00014346057,0.0013865479,0.015258735,0.002553758,0.6130753,0.013838795,0.35040092],"study_design_scores_gemma":[0.000055498593,0.0002441617,0.0013963294,0.00025325007,0.000030396859,0.00027589075,0.0007440009,0.07628122,0.0024299119,0.8173201,0.10092009,0.000049105485],"about_ca_topic_score_codex":0.0017344028,"about_ca_topic_score_gemma":0.0018235102,"teacher_disagreement_score":0.003086119,"about_ca_system_score_codex":0.0007512605,"about_ca_system_score_gemma":0.0005832115,"threshold_uncertainty_score":0.010482013},"labels":[],"label_agreement":null},{"id":"W1548867233","doi":"10.1007/978-3-540-30115-8_33","title":"Sparse Distributed Memories for On-Line Value-Based Reinforcement Learning","year":2004,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":52,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Reinforcement learning; A priori and a posteriori; Bellman equation; Line (geometry); Function (biology); Scheme (mathematics); Dynamic random-access memory; Distributed computing; Artificial intelligence; Mathematical optimization; Semiconductor memory; Computer hardware","score_opus":0.030305597366027073,"score_gpt":0.26716725850461265,"score_spread":0.2368616611385856,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1548867233","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01083771,0.00028355184,0.9847698,0.00015102932,0.00008832184,0.000037183374,0.000046491677,0.00045662135,0.0033293713],"genre_scores_gemma":[0.81123227,0.0004382778,0.178263,0.0001881378,0.00010862381,0.00027555847,0.00016826698,0.000112538226,0.009213357],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997298,0.000069883616,0.000018828088,0.000059242524,0.000081517384,0.000040721956],"domain_scores_gemma":[0.99909234,0.0005368069,0.00006175548,0.00013372785,0.00013768353,0.000037638645],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00051289523,0.0006754221,0.0011055782,0.0002527729,0.00029811816,0.00088452426,0.0015154385,0.0009264607,0.006860473],"category_scores_gemma":[0.002233867,0.0003885601,0.00035997847,0.00052508287,0.0006667335,0.0011238834,0.0011746667,0.0021624896,0.000712018],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021479395,0.00014896643,0.00024615575,0.00013952897,0.000058582737,0.00006860024,0.00007093942,0.7213832,0.0039305976,0.059678137,0.0042894906,0.20977102],"study_design_scores_gemma":[0.000017235492,0.000025111573,0.000032635147,0.0000048666107,0.000005104475,0.000011168268,0.000003840272,0.97245854,0.00055550516,0.026448067,0.00043396564,0.000003915335],"about_ca_topic_score_codex":0.0019570617,"about_ca_topic_score_gemma":0.0035358593,"teacher_disagreement_score":0.006860473,"about_ca_system_score_codex":0.0006177497,"about_ca_system_score_gemma":0.0005345953,"threshold_uncertainty_score":0.02295053},"labels":[],"label_agreement":null},{"id":"W1552645109","doi":"10.1184/r1/6558128","title":"Plays as Effective Multiagent Plans Enabling Opponent-Adaptive Play Selection","year":2018,"lang":"en","type":"article","venue":"Figshare","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":70,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"U.S. Air Force","keywords":"Computer science; Adversarial system; Adversary; Robot; Set (abstract data type); Adaptation (eye); Domain (mathematical analysis); Focus (optics); Plan (archaeology); Artificial intelligence; Human–computer interaction; Selection (genetic algorithm); Multi-agent system; Computer security","score_opus":0.03495788689315934,"score_gpt":0.27230603584848434,"score_spread":0.237348148955325,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1552645109","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043473918,0.00018532468,0.9459595,0.00022703382,0.000068539084,0.00016385663,0.00009144067,0.0015453581,0.008285025],"genre_scores_gemma":[0.7453131,0.00017944294,0.24716732,0.0001323243,0.00003202349,0.0002988401,0.00015819618,0.00018557729,0.006533203],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996284,0.00013332615,0.00001802882,0.000062731255,0.00011297011,0.000044514352],"domain_scores_gemma":[0.99924624,0.00034188,0.00011107007,0.00012488823,0.00007387383,0.00010215542],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00070262223,0.0006253327,0.00028968707,0.00037598354,0.00036247587,0.0008419577,0.0009847606,0.00058055023,0.0038418465],"category_scores_gemma":[0.0022543995,0.0003843375,0.00035571828,0.00021532418,0.0013448655,0.0011467673,0.0016330448,0.0010680299,0.0006020858],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003982269,0.00020574368,0.0020232582,0.00015153483,0.00007899174,0.00054759806,0.00065364194,0.6538045,0.022830518,0.18715549,0.0050747353,0.12707573],"study_design_scores_gemma":[0.000042763684,0.00012387603,0.00038038383,0.000019805386,0.00001748773,0.00007693493,0.00008012321,0.92787635,0.0045925495,0.060682952,0.006091495,0.000015247513],"about_ca_topic_score_codex":0.0010764392,"about_ca_topic_score_gemma":0.0018461654,"teacher_disagreement_score":0.0038418465,"about_ca_system_score_codex":0.00045868463,"about_ca_system_score_gemma":0.0005465858,"threshold_uncertainty_score":0.0128522515},"labels":[],"label_agreement":null},{"id":"W1552684655","doi":"","title":"Using linear programming for Bayesian exploration in Markov decision processes","year":2007,"lang":"en","type":"article","venue":"International Joint Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Markov decision process; Computer science; Bellman equation; Reinforcement learning; Mathematical optimization; Markov process; Linear programming; Machine learning; Representation (politics); Artificial intelligence; Markov chain; Key (lock); Bayesian probability; Partially observable Markov decision process; Markov model; Algorithm; Mathematics","score_opus":0.2143689443380605,"score_gpt":0.3942220206217933,"score_spread":0.1798530762837328,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1552684655","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0034999752,0.0004855718,0.9937435,0.00046067682,0.000016774407,0.000030407105,0.000035031353,0.00013075247,0.0015972736],"genre_scores_gemma":[0.45711496,0.0020444638,0.5301758,0.00061985856,0.00026521643,0.0011767809,0.0003608175,0.00032834,0.007913793],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.997115,0.0019290268,0.000077979166,0.00028391212,0.00037714824,0.00021692262],"domain_scores_gemma":[0.98595864,0.012835914,0.00051731814,0.00019203762,0.0003100446,0.0001860604],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054294486,0.0017465275,0.0024312125,0.0011694283,0.0007312061,0.0020017351,0.0019286126,0.0021008032,0.00510543],"category_scores_gemma":[0.019224178,0.0012300144,0.0012715044,0.001840999,0.0027305214,0.0028836026,0.0024246064,0.0038770903,0.0007216192],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000054432283,0.000060576913,0.00036294642,0.0001034416,0.00004078874,0.000046715355,0.00010031881,0.8628524,0.00015966497,0.11833105,0.0007718895,0.017115746],"study_design_scores_gemma":[0.00001130209,0.000011046827,0.000024809278,0.00001110201,0.000003652657,0.0000041737494,0.0000059935883,0.9287184,0.000046599896,0.070894994,0.0002624415,0.00000545611],"about_ca_topic_score_codex":0.008297607,"about_ca_topic_score_gemma":0.008343421,"teacher_disagreement_score":0.008297607,"about_ca_system_score_codex":0.0029588805,"about_ca_system_score_gemma":0.0025166066,"threshold_uncertainty_score":0.028714001},"labels":[],"label_agreement":null},{"id":"W1554247770","doi":"10.5555/1838206.1838354","title":"Cultivating desired behaviour: policy teaching via environment-dynamics tweaks","year":2010,"lang":"en","type":"article","venue":"ePrints Soton (University of Southampton)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Dynamics (music); Order (exchange); Function (biology); Control (management); Balance (ability); Ideal (ethics); System dynamics; Process management; Management science; Artificial intelligence; Engineering; Business; Psychology; Pedagogy; Political science","score_opus":0.007248548219061583,"score_gpt":0.19957262442598298,"score_spread":0.19232407620692138,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1554247770","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22379172,0.00035679078,0.75534964,0.0021229573,0.00008337722,0.00018631661,0.00003999712,0.0005777371,0.01749146],"genre_scores_gemma":[0.92035085,0.00023230896,0.07675363,0.00017332278,0.000022770371,0.00012805946,0.000019554755,0.000046670553,0.0022727102],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992717,0.00043540038,0.00002270294,0.00011831724,0.00010064109,0.000051200128],"domain_scores_gemma":[0.9955147,0.00323191,0.0005026739,0.00039274318,0.00016784374,0.00019010321],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014069679,0.0005208925,0.00034490286,0.00025619834,0.00032052156,0.00096984435,0.0007839582,0.0007779085,0.002716861],"category_scores_gemma":[0.01034507,0.0002475612,0.00020955656,0.00020000449,0.0012437038,0.0017830094,0.0012578095,0.0012412262,0.0003370463],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040261066,0.00136431,0.008742037,0.0005774096,0.00013252383,0.0003955033,0.0022341367,0.58609635,0.023987478,0.1765804,0.002525509,0.19696175],"study_design_scores_gemma":[0.00018141414,0.00051511306,0.0023944145,0.00008856094,0.000052762563,0.00013264203,0.00045941814,0.8476682,0.008634912,0.12933369,0.010484806,0.000054104734],"about_ca_topic_score_codex":0.0006788591,"about_ca_topic_score_gemma":0.001271763,"teacher_disagreement_score":0.002716861,"about_ca_system_score_codex":0.000506444,"about_ca_system_score_gemma":0.00078145723,"threshold_uncertainty_score":0.009088814},"labels":[],"label_agreement":null},{"id":"W1554366315","doi":"10.1023/a:1022145020786","title":"Approximate Gradient Methods in Policy-Space Optimization of Markov Reward Processes","year":2003,"lang":"en","type":"article","venue":"Discrete Event Dynamic Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":52,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Convergence (economics); Variance (accounting); Markov chain; Computer science; Markov process; Gradient descent; Path (computing); Set (abstract data type); Mathematics; Process (computing); Mathematical optimization; Algorithm; Artificial intelligence; Statistics; Artificial neural network","score_opus":0.012337159850539526,"score_gpt":0.3126734131687594,"score_spread":0.3003362533182199,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1554366315","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010953435,0.0010283438,0.9852496,0.0006218191,0.000084750136,0.000043710254,0.000035823436,0.00015682126,0.0018257117],"genre_scores_gemma":[0.72626597,0.0015258052,0.2616753,0.0003154683,0.00023832536,0.00054050505,0.00021573552,0.00032884727,0.008893954],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9987594,0.0007455999,0.000052479445,0.00010575531,0.00022217902,0.00011458521],"domain_scores_gemma":[0.99141616,0.007359692,0.000307921,0.0001682218,0.00052041886,0.00022771479],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004817448,0.0012555681,0.0027991338,0.0011624896,0.0007666865,0.0018106377,0.0019682588,0.0027034816,0.002466081],"category_scores_gemma":[0.019708246,0.0015434354,0.00074099266,0.0012426878,0.0026755407,0.0024863759,0.0024564539,0.0027696919,0.0003386377],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000056971006,0.000033700875,0.00020089246,0.00006723974,0.000030962077,0.000020448317,0.000045978755,0.9444474,0.00010780404,0.04538145,0.0005482123,0.009059083],"study_design_scores_gemma":[0.00000816884,0.000004806346,0.000015954374,0.000003879809,0.0000023178623,0.0000012753052,0.0000022638408,0.98872656,0.000022813278,0.011120452,0.00008958683,0.0000020333969],"about_ca_topic_score_codex":0.018871395,"about_ca_topic_score_gemma":0.010996341,"teacher_disagreement_score":0.018871395,"about_ca_system_score_codex":0.0025392203,"about_ca_system_score_gemma":0.003166313,"threshold_uncertainty_score":0.03752309},"labels":[],"label_agreement":null},{"id":"W1555338578","doi":"10.5555/1838206.1838401","title":"Using bisimulation for policy transfer in MDPs","year":2010,"lang":"en","type":"article","venue":"Adaptive Agents and Multi-Agents Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Markov decision process; Computer science; Bisimulation; Artificial intelligence; Markov process; Work (physics); Theoretical computer science; Mathematics","score_opus":0.12857996272755987,"score_gpt":0.35545820968810005,"score_spread":0.22687824696054018,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1555338578","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0070414506,0.00021767002,0.98847896,0.00025097094,0.000044099623,0.00007835201,0.000055729073,0.00029013355,0.0035426056],"genre_scores_gemma":[0.65099955,0.0008191057,0.3404536,0.00042765285,0.00010781856,0.0013405356,0.00033171423,0.00044518377,0.0050749425],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99774855,0.0012324632,0.00014335536,0.00035822438,0.00034837294,0.00016908409],"domain_scores_gemma":[0.99052167,0.0075329286,0.00061610574,0.00053942314,0.00050716265,0.00028270014],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004637215,0.002022827,0.0022145156,0.0013044296,0.00089016306,0.00172718,0.0020620665,0.0022836775,0.0074841646],"category_scores_gemma":[0.018574342,0.0011034906,0.0015032444,0.0010302804,0.0026378501,0.003376481,0.004161487,0.0034203758,0.0010740934],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000045220415,0.000033197015,0.00015392744,0.00007328436,0.000032157674,0.000039419312,0.000058558253,0.9152627,0.00028387073,0.07266043,0.0003094513,0.011047772],"study_design_scores_gemma":[0.000016420345,0.000021776379,0.00001312821,0.000013279085,0.000005537146,0.0000063818766,0.0000058465685,0.9485861,0.00017486628,0.05065836,0.0004915995,0.0000066071043],"about_ca_topic_score_codex":0.0045451494,"about_ca_topic_score_gemma":0.0033332317,"teacher_disagreement_score":0.0074841646,"about_ca_system_score_codex":0.0024286497,"about_ca_system_score_gemma":0.0025147903,"threshold_uncertainty_score":0.02503705},"labels":[],"label_agreement":null},{"id":"W1556824961","doi":"10.1007/3-540-45622-8_16","title":"Learning Options in Reinforcement Learning","year":2002,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":270,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Psychology; Artificial intelligence","score_opus":0.021089761123276374,"score_gpt":0.2445734416842048,"score_spread":0.22348368056092843,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1556824961","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014524392,0.0062356535,0.91782737,0.0015067187,0.0003095269,0.00003816838,0.000064402906,0.00028388377,0.05920988],"genre_scores_gemma":[0.68233055,0.0060686017,0.25105098,0.00041751866,0.0004628431,0.00028971717,0.00019398786,0.0001673007,0.05901855],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99967146,0.0001679115,0.000014747866,0.00004713654,0.00007814249,0.000020586576],"domain_scores_gemma":[0.9990978,0.000738562,0.000029392779,0.000057759564,0.000046858924,0.00002963823],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00081407645,0.00063182006,0.00068008946,0.00028850796,0.00026108974,0.001121903,0.00079766015,0.0008988484,0.007125658],"category_scores_gemma":[0.0029137875,0.0003402363,0.00039167702,0.0004951528,0.0015274689,0.0025659525,0.00080714625,0.002286638,0.0008303144],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000042236385,0.000044846573,0.00020779905,0.00013244637,0.000026238102,0.0000570968,0.00011979473,0.07452527,0.00046489394,0.78527963,0.0046382886,0.13446146],"study_design_scores_gemma":[0.000014624526,0.000021955271,0.00006504189,0.000031062773,0.000007649739,0.000027692908,0.000017408593,0.15377645,0.00031557315,0.83972436,0.0059897746,0.000008328362],"about_ca_topic_score_codex":0.0005394966,"about_ca_topic_score_gemma":0.00062017696,"teacher_disagreement_score":0.007125658,"about_ca_system_score_codex":0.00065483956,"about_ca_system_score_gemma":0.00030640632,"threshold_uncertainty_score":0.023837686},"labels":[],"label_agreement":null},{"id":"W15601695","doi":"10.1128/jb.187.1.114-124.2005","title":"Autolysis of Lactococcus lactis is increased upon D-alanine depletion of peptidoglycan and lipoteichoic acids.","year":2005,"lang":"en","type":"article","venue":"PubMed","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Kernel (algebra); Kernel method; Mathematical optimization; Algorithm; Artificial intelligence; Mathematics; Discrete mathematics","score_opus":0.01333494179846988,"score_gpt":0.21369268491157492,"score_spread":0.20035774311310503,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W15601695","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9939254,0.0017975181,0.0019491266,0.00010291576,0.000030041447,0.000028269813,0.00055268593,0.00017643512,0.0014375715],"genre_scores_gemma":[0.9945657,0.0005441162,0.0018670287,0.000084464366,0.00000888777,0.0000363794,0.0011212304,0.000040718598,0.0017314188],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99956197,0.00005862955,0.000057992154,0.000070284244,0.00015333564,0.000097699565],"domain_scores_gemma":[0.9996216,0.000067354675,0.00011875894,0.000052875712,0.000057785717,0.00008154923],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00014884715,0.0006757859,0.0005074798,0.0003147712,0.000111027635,0.00051374466,0.00021287335,0.00034613413,0.00085251307],"category_scores_gemma":[0.00023496887,0.00018308792,0.00040455608,0.00027550833,0.00020934845,0.00026308603,0.0005955515,0.00052196905,0.0005569766],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006990101,0.0000136911685,0.00026916503,0.0000191779,0.0000042310244,0.00006284006,0.000010429696,0.000009916313,0.99896634,0.0000127293615,0.000011853511,0.0005497044],"study_design_scores_gemma":[0.000006262045,0.00020305142,0.009957822,0.000006178734,0.000015727805,0.0005391013,0.00004439822,0.0003051595,0.9878456,0.00002564739,0.0010441765,0.000006905285],"about_ca_topic_score_codex":0.0009419,"about_ca_topic_score_gemma":0.0005834789,"teacher_disagreement_score":0.0009419,"about_ca_system_score_codex":0.00034198747,"about_ca_system_score_gemma":0.0002188572,"threshold_uncertainty_score":0.002851963},"labels":[],"label_agreement":null},{"id":"W1561485809","doi":"10.1007/978-3-540-70829-2_11","title":"The Concept of Opposition and Its Use in Q-Learning and Q(λ) Techniques","year":2008,"lang":"en","type":"book-chapter","venue":"Studies in computational intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Opposition (politics); Q-learning; Artificial intelligence; TRACE (psycholinguistics); Mathematical optimization; Mathematics","score_opus":0.09245566855671966,"score_gpt":0.3442205430760504,"score_spread":0.2517648745193307,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1561485809","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0022828067,0.0027804908,0.96955854,0.0011984248,0.00038701307,0.000026741312,0.000023211463,0.00007325439,0.023669483],"genre_scores_gemma":[0.33382082,0.0067344015,0.6359086,0.0015571368,0.0014573884,0.00051891565,0.00007810693,0.00023984566,0.019684799],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99759966,0.0014121011,0.00010436134,0.00023553948,0.0005668092,0.00008151026],"domain_scores_gemma":[0.99300826,0.006010381,0.0002439104,0.00030454568,0.0003161997,0.00011677263],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003440209,0.00081156625,0.0011336624,0.001100267,0.0009788582,0.0020840613,0.0019676453,0.0018597072,0.0049222675],"category_scores_gemma":[0.0118029,0.00053530734,0.0010401,0.00251849,0.008822698,0.0050633606,0.0030698164,0.005611728,0.0010490712],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001721151,0.000010657864,0.000047754114,0.000058164187,0.0000068872346,0.00001909732,0.0000623026,0.0041658394,0.00022429222,0.97260416,0.0010683365,0.021715166],"study_design_scores_gemma":[0.000013636171,0.0000270886,0.000052429757,0.00003184866,0.00000777199,0.000058855334,0.000019768457,0.03212463,0.00023968394,0.9583342,0.009077661,0.000012458278],"about_ca_topic_score_codex":0.0008000039,"about_ca_topic_score_gemma":0.0005011817,"teacher_disagreement_score":0.0049222675,"about_ca_system_score_codex":0.00093361613,"about_ca_system_score_gemma":0.0006150719,"threshold_uncertainty_score":0.018193841},"labels":[],"label_agreement":null},{"id":"W1562324652","doi":"10.1111/coin.12060","title":"Concurrent Individual And Social Learning In Robot Teams","year":2015,"lang":"en","type":"article","venue":"Computational Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institute for Christian Studies; University of Toronto","funders":"","keywords":"Robot; Computer science; Team learning; Relation (database); Human–computer interaction; Artificial intelligence; Knowledge management; Cooperative learning; Psychology; Teaching method; Open learning","score_opus":0.07697776732268413,"score_gpt":0.3196348699339635,"score_spread":0.24265710261127937,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1562324652","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.49870962,0.000677151,0.4672196,0.001554891,0.00010339041,0.0000887202,0.000027023549,0.0002178762,0.03140172],"genre_scores_gemma":[0.98779243,0.000088125504,0.010495788,0.000028112476,0.000018678931,0.0000402804,0.0000058506084,0.0000075298885,0.0015231682],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989612,0.0005009288,0.000028696857,0.00015261112,0.00023448344,0.0001220059],"domain_scores_gemma":[0.99731976,0.0016865035,0.00033132764,0.00018748043,0.00017985768,0.00029499008],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016980077,0.0002443318,0.00031922004,0.00030714922,0.00065290206,0.0011052055,0.0007379428,0.0006144251,0.0013427038],"category_scores_gemma":[0.0062453416,0.00019997667,0.00027996372,0.000203368,0.0023869982,0.0014799294,0.0016047239,0.00064833893,0.0001339286],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018129082,0.0002143562,0.00434945,0.00011896237,0.00008872129,0.00032335936,0.001774184,0.47988984,0.0035791104,0.4306077,0.0011948956,0.07767802],"study_design_scores_gemma":[0.00006396185,0.00010933555,0.0014232585,0.000015621368,0.00001896985,0.00005616958,0.0003065245,0.6232296,0.00092915224,0.3711036,0.0027234268,0.000020447062],"about_ca_topic_score_codex":0.0022807017,"about_ca_topic_score_gemma":0.0016744001,"teacher_disagreement_score":0.0022807017,"about_ca_system_score_codex":0.00086619385,"about_ca_system_score_gemma":0.0010406937,"threshold_uncertainty_score":0.008979976},"labels":[],"label_agreement":null},{"id":"W1565394148","doi":"","title":"Solving POMDPs with continuous or large discrete observation spaces","year":2005,"lang":"en","type":"article","venue":"Discovery Research Portal (University of Dundee)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":98,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; University of Toronto","funders":"","keywords":"Discretization; Markov decision process; Partially observable Markov decision process; Partition (number theory); Computer science; A priori and a posteriori; Observable; Task (project management); Domain (mathematical analysis); Mathematical optimization; Space (punctuation); Markov chain; Markov process; Theoretical computer science; Mathematics; Markov model; Machine learning","score_opus":0.046985247959016346,"score_gpt":0.29276817378004755,"score_spread":0.2457829258210312,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1565394148","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005089976,0.00008372937,0.9935992,0.00009713573,0.000017264449,0.00004536572,0.0000374978,0.00019742576,0.00083232817],"genre_scores_gemma":[0.2496398,0.00032140734,0.74793804,0.00008365346,0.00004572331,0.0003099312,0.00018993694,0.000068814414,0.001402748],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99910396,0.0002963775,0.000068895606,0.0002149946,0.00021402675,0.00010179643],"domain_scores_gemma":[0.9976673,0.0017966059,0.00018891616,0.00018210648,0.00009379404,0.000071279246],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014021217,0.0011154106,0.0010964532,0.00033655536,0.00068327703,0.0011975471,0.0014358056,0.0013928522,0.002788364],"category_scores_gemma":[0.004071516,0.00073531945,0.0011489689,0.00052587246,0.0013096874,0.0015405774,0.0017330818,0.0025595108,0.0003233403],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000042844993,0.000054389777,0.00033915637,0.000169183,0.000043477063,0.00009119586,0.00007743819,0.94012624,0.0014496812,0.029725324,0.0005001748,0.027380874],"study_design_scores_gemma":[0.000022834884,0.000023647148,0.000054248805,0.000010873388,0.000008636458,0.000020223932,0.00001395915,0.97388214,0.00089935673,0.02389197,0.0011648847,0.0000071853497],"about_ca_topic_score_codex":0.003506643,"about_ca_topic_score_gemma":0.004015807,"teacher_disagreement_score":0.003506643,"about_ca_system_score_codex":0.0007405536,"about_ca_system_score_gemma":0.0015446726,"threshold_uncertainty_score":0.009328008},"labels":[],"label_agreement":null},{"id":"W1565405003","doi":"10.1007/978-3-540-68825-9_13","title":"Point-Based Planning for Predictive State Representations","year":2008,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Observability; Computer science; Partially observable Markov decision process; Representation (politics); Observable; Markov decision process; Automated planning and scheduling; State (computer science); Point (geometry); Mathematical optimization; Artificial intelligence; Markov chain; Algorithm; Theoretical computer science; Markov process; Machine learning; Markov model; Mathematics; Applied mathematics","score_opus":0.02775704980736581,"score_gpt":0.2808852751791834,"score_spread":0.25312822537181756,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1565405003","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032672782,0.00032577218,0.9935814,0.00004055578,0.00002559224,0.000020878617,0.000042682484,0.0003549167,0.0023408693],"genre_scores_gemma":[0.4335136,0.0013867959,0.5544614,0.000085625434,0.000059201142,0.000368449,0.00044758187,0.00026207388,0.009415178],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99968624,0.000072775954,0.000023128094,0.00006708343,0.00011833044,0.000032480755],"domain_scores_gemma":[0.999438,0.0003896409,0.000027026375,0.00006793872,0.00005924297,0.000018154724],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006596077,0.0007568788,0.0011873551,0.0005259905,0.00040290595,0.00092607155,0.0017455302,0.001048721,0.005281249],"category_scores_gemma":[0.0021244679,0.00086985674,0.00080515206,0.0010932877,0.0009535367,0.0015668925,0.001251586,0.0017539631,0.0008728281],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000061790866,0.00002347082,0.00009396184,0.000109710556,0.000024483701,0.0000522758,0.00006720454,0.8385354,0.0013078626,0.064186305,0.0015134333,0.09402411],"study_design_scores_gemma":[0.0000047303683,0.000011252805,0.000020753743,0.000008387549,0.000004105859,0.0000089590185,0.000004332476,0.96398586,0.00038366302,0.03496114,0.00060222397,0.0000046101813],"about_ca_topic_score_codex":0.0060825017,"about_ca_topic_score_gemma":0.006154992,"teacher_disagreement_score":0.0060825017,"about_ca_system_score_codex":0.0008395631,"about_ca_system_score_gemma":0.0007012503,"threshold_uncertainty_score":0.017667532},"labels":[],"label_agreement":null},{"id":"W1569513695","doi":"10.1007/978-3-540-24840-8_30","title":"Multi-attribute Decision Making in a Complex Multiagent Environment Using Reinforcement Learning with Selective Perception","year":2004,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Reinforcement learning; Computer science; Task (project management); Perception; Artificial intelligence; Machine learning; Human–computer interaction","score_opus":0.03624636363657711,"score_gpt":0.27589725601734344,"score_spread":0.23965089238076634,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1569513695","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0777959,0.0001437408,0.91966414,0.00019932109,0.00003884814,0.00005764701,0.000016890408,0.00020762712,0.0018758917],"genre_scores_gemma":[0.8469521,0.00010523712,0.15151465,0.0000594824,0.000026223284,0.00011298298,0.000025861606,0.000028283968,0.0011751448],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993401,0.00027417988,0.00004123724,0.00012702873,0.00015363253,0.00006380995],"domain_scores_gemma":[0.9971909,0.0020248457,0.00019912598,0.00016607369,0.0002442734,0.00017482492],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019404561,0.0006290954,0.0010566018,0.00027831263,0.00047299554,0.0011181707,0.0012734678,0.00093716464,0.0014924564],"category_scores_gemma":[0.004299837,0.000471988,0.00067780144,0.00039629856,0.0010764658,0.001540647,0.0013533004,0.0015909184,0.00015611017],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019661535,0.00020493622,0.0013210817,0.000071105394,0.00013115891,0.00014606252,0.00020399655,0.915617,0.0041968334,0.016084233,0.0006060134,0.061220873],"study_design_scores_gemma":[0.000012609304,0.000028384113,0.00010512259,0.0000031130903,0.000009525411,0.000011194345,0.00000823925,0.993067,0.00043862735,0.006236064,0.00007498507,0.000005133167],"about_ca_topic_score_codex":0.0027289274,"about_ca_topic_score_gemma":0.0025117944,"teacher_disagreement_score":0.0027289274,"about_ca_system_score_codex":0.0007486199,"about_ca_system_score_gemma":0.0007400093,"threshold_uncertainty_score":0.010262251},"labels":[],"label_agreement":null},{"id":"W1591992921","doi":"10.48550/arxiv.1207.4114","title":"Metrics for Finite Markov Decision Processes","year":2012,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Markov decision process; Reinforcement learning; Metric (unit); Markov process; Computer science; Markov kernel; Markov chain; Aggregate (composite); Bellman equation; Mathematical optimization; Q-learning; Similarity (geometry); Variable-order Markov model; Artificial intelligence; Markov model; Machine learning; Mathematics","score_opus":0.07973034372712487,"score_gpt":0.2043299711430501,"score_spread":0.12459962741592523,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1591992921","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016234357,0.0034441086,0.9718736,0.000770467,0.0001696732,0.00012533941,0.0005367389,0.0002824379,0.0065633403],"genre_scores_gemma":[0.56793255,0.0036668757,0.41955397,0.0005157906,0.0005000349,0.000928324,0.0018645378,0.00029696833,0.0047408636],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9924541,0.0028349182,0.00089249416,0.001228252,0.0021976281,0.00039260025],"domain_scores_gemma":[0.97412103,0.01680393,0.0032593012,0.0017257107,0.0026930515,0.0013970807],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052542486,0.001948799,0.0014264149,0.003596262,0.0011309641,0.0035195583,0.00174653,0.0020408623,0.0040802537],"category_scores_gemma":[0.03643912,0.0005769226,0.001212018,0.0026849422,0.002684503,0.007987447,0.0033658224,0.003338325,0.0007347395],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006225424,0.00004498997,0.0012799795,0.00023589093,0.00008051059,0.00010088357,0.00022024052,0.104197204,0.0009841493,0.85556597,0.0023100763,0.03491783],"study_design_scores_gemma":[0.000008714562,0.0000761536,0.00033416285,0.000056426397,0.000017535634,0.00007910645,0.000033759392,0.23250194,0.00043442546,0.76106715,0.005363937,0.000026746198],"about_ca_topic_score_codex":0.0019690972,"about_ca_topic_score_gemma":0.0012556488,"teacher_disagreement_score":0.0052542486,"about_ca_system_score_codex":0.0034770244,"about_ca_system_score_gemma":0.0015082253,"threshold_uncertainty_score":0.027787507},"labels":[],"label_agreement":null},{"id":"W1595080983","doi":"10.1109/acc.2015.7171887","title":"Finite state approximations of Markov decision processes with general state and action spaces","year":2015,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Markov decision process; Action (physics); Finite state; Markov process; Complement (music); State space; Mathematical optimization; Markov chain; Applied mathematics; Mathematics; State (computer science); Partially observable Markov decision process; Markov model; Stochastic process; Markov kernel; Approximations of π; Computer science; Variable-order Markov model; Algorithm","score_opus":0.034492949131916716,"score_gpt":0.2734357435885215,"score_spread":0.2389427944566048,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1595080983","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043618258,0.0005057452,0.9526401,0.00029550822,0.000044072123,0.000028439094,0.00010594972,0.00012644728,0.0026355172],"genre_scores_gemma":[0.8809926,0.0006758776,0.11446505,0.000099414625,0.000060001963,0.0001599709,0.0003084342,0.000056170546,0.0031825257],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99812967,0.0008262715,0.00011225161,0.0003137872,0.0004513062,0.00016680664],"domain_scores_gemma":[0.98931295,0.00836774,0.0010406934,0.0006216416,0.00041203108,0.00024497733],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029963583,0.0009081318,0.0014457578,0.0007057815,0.0004554502,0.0018839523,0.0017434285,0.0013139065,0.001808957],"category_scores_gemma":[0.015254287,0.00073903886,0.0012215031,0.0007917273,0.002272935,0.0023954269,0.0014277895,0.0028032134,0.00024210471],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000031492666,0.000015080014,0.00023568896,0.000029166038,0.000017775243,0.000037391368,0.000044644057,0.9227804,0.00016900247,0.07403399,0.00009205625,0.0025133202],"study_design_scores_gemma":[0.0000037223083,0.0000060160755,0.00003398747,0.000005679438,0.000002423081,0.0000046130194,0.0000042816628,0.97131604,0.000060470884,0.02844179,0.00011788687,0.0000030545552],"about_ca_topic_score_codex":0.010560455,"about_ca_topic_score_gemma":0.0072072833,"teacher_disagreement_score":0.010560455,"about_ca_system_score_codex":0.0028416058,"about_ca_system_score_gemma":0.0017760297,"threshold_uncertainty_score":0.020997941},"labels":[],"label_agreement":null},{"id":"W1600046456","doi":"","title":"Off-Policy Temporal Difference Learning with Function Approximation","year":2001,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":254,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Temporal difference learning; Computer science; Function approximation; Function (biology); Artificial intelligence; Reinforcement learning; Artificial neural network","score_opus":0.01561973101356947,"score_gpt":0.23147654686657615,"score_spread":0.21585681585300667,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1600046456","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003496256,0.00015007531,0.9932249,0.00018401981,0.00006489106,0.000067299414,0.000027713686,0.0004195545,0.0023651957],"genre_scores_gemma":[0.42620316,0.0002563175,0.5633009,0.0007245957,0.000117612195,0.000512255,0.00027849717,0.00034073184,0.008266045],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99861705,0.00037096837,0.000075674725,0.0002993096,0.00044912976,0.00018785326],"domain_scores_gemma":[0.9960349,0.002621733,0.00025240387,0.000370947,0.00052080496,0.00019915367],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032946998,0.0011414422,0.001641894,0.00073500245,0.0005181968,0.0014252552,0.0030690155,0.0022716518,0.0073075052],"category_scores_gemma":[0.012637295,0.0006367684,0.0008568828,0.00081968954,0.001867261,0.002441896,0.0029882907,0.0037695374,0.0014984009],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031348638,0.00023620602,0.0010608619,0.0001587575,0.00005696307,0.000112883936,0.0001649289,0.63166916,0.0020815164,0.11233979,0.0043057483,0.24749973],"study_design_scores_gemma":[0.000015242442,0.000033643035,0.000028978553,0.000009603483,0.0000028961879,0.000018359016,0.0000039648535,0.98594725,0.000417484,0.012816739,0.0007013529,0.000004523283],"about_ca_topic_score_codex":0.0036361874,"about_ca_topic_score_gemma":0.0023140195,"teacher_disagreement_score":0.0073075052,"about_ca_system_score_codex":0.0019566147,"about_ca_system_score_gemma":0.0022379148,"threshold_uncertainty_score":0.02444601},"labels":[],"label_agreement":null},{"id":"W1601777908","doi":"10.1007/3-540-36755-1_33","title":"Characterizing Markov Decision Processes","year":2002,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Markov decision process; Computer science; Randomness; Artificial intelligence; Markov chain; Machine learning; Markov process; Measure (data warehouse); Quality (philosophy); Mathematical optimization; Decision problem; Algorithm; Data mining; Mathematics; Statistics","score_opus":0.019528231396003598,"score_gpt":0.243811645501996,"score_spread":0.2242834141059924,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1601777908","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.046600644,0.0006436036,0.9331105,0.00066349,0.00007489733,0.000077471625,0.00048049647,0.0002958258,0.018052999],"genre_scores_gemma":[0.88033664,0.00138192,0.092585325,0.0003022968,0.00027753148,0.0003822469,0.0018100054,0.00027580853,0.022648226],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99874425,0.00048839854,0.00006617873,0.00026678803,0.00029634705,0.00013795764],"domain_scores_gemma":[0.98886,0.0088866735,0.0006867629,0.00058422156,0.00064882834,0.0003335259],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017844662,0.0010365774,0.0012691161,0.0012172619,0.0006365254,0.0026088506,0.001100591,0.0012483597,0.0070390524],"category_scores_gemma":[0.015178638,0.00075222296,0.0010522262,0.0013798861,0.001130226,0.002566985,0.0016770437,0.0021537042,0.00086732995],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000048798127,0.00006620615,0.0013062457,0.00007258035,0.0000426594,0.000095605435,0.00016227523,0.091476694,0.0008396939,0.88324857,0.0023433636,0.020297334],"study_design_scores_gemma":[0.000007279023,0.000012347861,0.00019064314,0.000012653696,0.00000985703,0.000028382981,0.000018028353,0.3178175,0.00024621974,0.6806193,0.0010301864,0.000007557935],"about_ca_topic_score_codex":0.0017220876,"about_ca_topic_score_gemma":0.0012901741,"teacher_disagreement_score":0.0070390524,"about_ca_system_score_codex":0.0016898183,"about_ca_system_score_gemma":0.0010379624,"threshold_uncertainty_score":0.023547947},"labels":[],"label_agreement":null},{"id":"W1602154927","doi":"10.48550/arxiv.1301.2343","title":"Planning by Prioritized Sweeping with Small Backups","year":2013,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Business; Computer science; Process management","score_opus":0.07880461453518656,"score_gpt":0.18244058612788147,"score_spread":0.10363597159269491,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1602154927","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0633699,0.00022097467,0.93112254,0.00022490235,0.00007593296,0.000087949804,0.000095490104,0.0020469842,0.002755309],"genre_scores_gemma":[0.764391,0.00012303902,0.23305272,0.00009567505,0.000027591781,0.00015866167,0.000121781064,0.000164031,0.0018654246],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99948645,0.000119661636,0.000039688177,0.00013578073,0.00013870485,0.00007969283],"domain_scores_gemma":[0.99835217,0.0008128834,0.00013924311,0.00042916415,0.00012137012,0.00014514802],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009177895,0.0007743019,0.0008364636,0.00040780398,0.00045983927,0.0007815887,0.0014618955,0.00069927756,0.0047287443],"category_scores_gemma":[0.0036728685,0.00048503306,0.0004498576,0.0003838275,0.0010274093,0.0014580674,0.0016164806,0.0015655004,0.0005646871],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008348111,0.00022887267,0.0014619108,0.00017740455,0.000062919695,0.00027034018,0.00024924296,0.7350071,0.019730063,0.029197045,0.002322451,0.21045786],"study_design_scores_gemma":[0.0000612866,0.00007983413,0.00012102941,0.000009894447,0.000011833264,0.000042327392,0.000022302926,0.9789473,0.0031338288,0.016538737,0.0010218081,0.000009983117],"about_ca_topic_score_codex":0.0025163586,"about_ca_topic_score_gemma":0.0028416947,"teacher_disagreement_score":0.0047287443,"about_ca_system_score_codex":0.00047405716,"about_ca_system_score_gemma":0.0011909334,"threshold_uncertainty_score":0.015819192},"labels":[],"label_agreement":null},{"id":"W161119602","doi":"10.1007/978-1-84628-758-9_9","title":"Reinforcement Agents for E-Learning Applications","year":2007,"lang":"en","type":"book-chapter","venue":"Advanced information and knowledge processing","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Process (computing); Set (abstract data type); Reinforcement; Error-driven learning; Human–computer interaction; Artificial intelligence; Punishment (psychology); Engineering","score_opus":0.029622643948588556,"score_gpt":0.30180862920811563,"score_spread":0.2721859852595271,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W161119602","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0038604161,0.021666382,0.7607265,0.0012810512,0.0010740886,0.00012423177,0.00017045533,0.002327637,0.20876917],"genre_scores_gemma":[0.107631095,0.022791788,0.46969467,0.0005714414,0.0005991842,0.0004029942,0.000488434,0.0005567674,0.3972636],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998691,0.000025675965,0.000006557372,0.000017570508,0.00007210947,0.000008964019],"domain_scores_gemma":[0.9998642,0.00006639013,0.0000067163132,0.000021628855,0.00003148619,0.000009620639],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001721789,0.00067668135,0.0004420461,0.0002709249,0.00022773737,0.0009668243,0.0007635214,0.0009947413,0.018278647],"category_scores_gemma":[0.00051576126,0.00023641945,0.00023064918,0.00055609655,0.0003846401,0.0011738832,0.0006434878,0.0012044654,0.00504337],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000031388783,0.00010140901,0.00008661961,0.000389055,0.000021341386,0.00010214376,0.00008169265,0.030915657,0.004240247,0.2186362,0.051820945,0.69357324],"study_design_scores_gemma":[0.000031432002,0.000059337282,0.00019943295,0.000229353,0.000019905718,0.00032383788,0.00005167357,0.12559451,0.0053363694,0.241939,0.62619203,0.000023134371],"about_ca_topic_score_codex":0.00062849163,"about_ca_topic_score_gemma":0.00086628593,"teacher_disagreement_score":0.018278647,"about_ca_system_score_codex":0.00040451117,"about_ca_system_score_gemma":0.00037443117,"threshold_uncertainty_score":0.061148226},"labels":[],"label_agreement":null},{"id":"W1672238326","doi":"10.1109/ccece.2015.7129412","title":"The residual gradient FACL algorithm for differential games","year":2015,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Fuzzy logic; Algorithm; Convergence (economics); Computer science; Reinforcement learning; Residual; Fuzzy control system; Adaptive neuro fuzzy inference system; Control theory (sociology); Artificial intelligence; Mathematics; Control (management)","score_opus":0.03577575525246496,"score_gpt":0.26973635512533206,"score_spread":0.2339605998728671,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1672238326","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0027647435,0.00014581754,0.9938822,0.00010406738,0.000035385066,0.00007100787,0.00001541725,0.00023477458,0.0027466402],"genre_scores_gemma":[0.44008538,0.00031153116,0.54838777,0.00032146802,0.00006749723,0.00079567597,0.00014412378,0.00014743848,0.009739004],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995277,0.0001589177,0.000025527175,0.00007741492,0.00015708526,0.000053441952],"domain_scores_gemma":[0.999401,0.0003006731,0.0000693175,0.00003698534,0.00015326445,0.000038777758],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011977194,0.0010988624,0.0012516637,0.0008134813,0.00048397479,0.000841129,0.0018401825,0.0011390949,0.0036867056],"category_scores_gemma":[0.0021994684,0.00037962606,0.0004679236,0.00043051236,0.0010724657,0.0008471955,0.0010699701,0.001345515,0.00087913324],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008534059,0.00007574224,0.00049882964,0.00015601261,0.000049092534,0.00008803481,0.00012767475,0.7795929,0.0019169906,0.06004096,0.0035211437,0.1538472],"study_design_scores_gemma":[0.00001600177,0.000023853758,0.000028235712,0.0000054941183,0.0000027144038,0.000014059457,0.0000041773856,0.99378926,0.00021916917,0.0049892073,0.0009032869,0.0000046144155],"about_ca_topic_score_codex":0.0060621435,"about_ca_topic_score_gemma":0.0038640406,"teacher_disagreement_score":0.0060621435,"about_ca_system_score_codex":0.0013724563,"about_ca_system_score_gemma":0.0015631886,"threshold_uncertainty_score":0.012333214},"labels":[],"label_agreement":null},{"id":"W1758031947","doi":"10.48550/arxiv.1206.3285","title":"Dyna-Style Planning with Linear Function Approximation and Prioritized Sweeping","year":2012,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":107,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Bellman equation; Function (biology); Limit (mathematics); Mathematical optimization; Function approximation; State (computer science); Linear approximation; Linear programming; Artificial intelligence; Algorithm; Mathematics; Nonlinear system; Artificial neural network","score_opus":0.0529436338934253,"score_gpt":0.18302947821939336,"score_spread":0.13008584432596806,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1758031947","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03115027,0.00016928023,0.9649688,0.00022304262,0.000024963194,0.000040524206,0.000055018656,0.000677253,0.002690889],"genre_scores_gemma":[0.79790306,0.0001407247,0.19786756,0.00011088364,0.000025681995,0.00014918663,0.00009955,0.000096185635,0.0036071779],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99939084,0.00024076414,0.000031780004,0.0001282652,0.00012236537,0.000085897875],"domain_scores_gemma":[0.9983981,0.0011093183,0.00012485868,0.000166153,0.00011117007,0.000090381836],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012754793,0.0008478198,0.00095174095,0.0004985373,0.00045736588,0.00095849996,0.001277655,0.001093393,0.0032949646],"category_scores_gemma":[0.0038626315,0.00062700815,0.00057473383,0.0006258675,0.001383964,0.0015038893,0.0014113857,0.0014129302,0.00037682682],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006698437,0.000030640287,0.0003396661,0.000037684535,0.000021315593,0.00004654857,0.000059832848,0.9609173,0.0006179055,0.017257806,0.00025305495,0.020351198],"study_design_scores_gemma":[0.000008228252,0.000016469936,0.000024275707,0.000002626411,0.0000027231342,0.000005666138,0.0000042622123,0.9918371,0.00021314934,0.0076904823,0.00019216316,0.00000278215],"about_ca_topic_score_codex":0.007459655,"about_ca_topic_score_gemma":0.006504701,"teacher_disagreement_score":0.007459655,"about_ca_system_score_codex":0.0010734361,"about_ca_system_score_gemma":0.0015090377,"threshold_uncertainty_score":0.014832437},"labels":[],"label_agreement":null},{"id":"W176737593","doi":"","title":"Planning and programming with first-order markov decision processes: insights and challenges","year":2001,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Markov decision process; Bellman equation; Computer science; Dynamic programming; Mathematical optimization; State space; Representation (politics); Markov process; Automated planning and scheduling; State (computer science); Function (biology); Markov chain; Mathematics; Artificial intelligence; Algorithm; Machine learning","score_opus":0.02615278811115168,"score_gpt":0.2471596119159763,"score_spread":0.22100682380482461,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W176737593","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009952836,0.024770752,0.93229216,0.01758158,0.0002876598,0.000049251383,0.00019161067,0.00012989309,0.014744308],"genre_scores_gemma":[0.50789845,0.06287065,0.41229263,0.0023089207,0.0032221945,0.0004902516,0.00041215224,0.00017782119,0.01032694],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99785006,0.0011574782,0.000087182205,0.0002601971,0.0005120393,0.00013301054],"domain_scores_gemma":[0.98746115,0.010819972,0.0005534985,0.00039274927,0.0004906333,0.00028195517],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042207935,0.0011119262,0.0016023635,0.0009804561,0.00082350965,0.003973485,0.0019434773,0.002245515,0.0028971585],"category_scores_gemma":[0.011924755,0.0010133759,0.0011291482,0.0019420414,0.0042654225,0.008801307,0.0023477015,0.005326637,0.00047823935],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000013987295,0.00004056818,0.000369557,0.00018611207,0.000026083255,0.000072689145,0.00015202981,0.055854023,0.00007069206,0.92300326,0.0019293266,0.018281652],"study_design_scores_gemma":[0.00000533737,0.000008591668,0.000082468265,0.00004212613,0.0000034534976,0.000019542198,0.000045284094,0.0977901,0.00003296537,0.8988727,0.0030897919,0.000007716197],"about_ca_topic_score_codex":0.0048247785,"about_ca_topic_score_gemma":0.0043925345,"teacher_disagreement_score":0.0048247785,"about_ca_system_score_codex":0.0022554756,"about_ca_system_score_gemma":0.002684784,"threshold_uncertainty_score":0.022322},"labels":[],"label_agreement":null},{"id":"W1787017840","doi":"10.1016/j.jmaa.2015.10.008","title":"Near optimality of quantized policies in stochastic control under weak continuity conditions","year":2015,"lang":"en","type":"preprint","venue":"Journal of Mathematical Analysis and Applications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Markov decision process; Discretization; Constructive; Markov process; Mathematical optimization; Optimal control; Stochastic control; Mathematics; Partially observable Markov decision process; Reduction (mathematics); Markov chain; Class (philosophy); Applied mathematics; Computer science; Process (computing); Mathematical analysis","score_opus":0.030012285172124363,"score_gpt":0.32465065521903924,"score_spread":0.2946383700469149,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1787017840","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15581366,0.0011121192,0.82592386,0.0028031075,0.000223601,0.00011812028,0.00022969385,0.00022246136,0.013553379],"genre_scores_gemma":[0.96336097,0.000685099,0.030547839,0.0003152641,0.00014865474,0.00015903171,0.00013453847,0.00010705024,0.004541661],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9976292,0.0011512447,0.00013669877,0.00037358236,0.00044212185,0.00026713422],"domain_scores_gemma":[0.97847646,0.017526291,0.0013939331,0.0006109077,0.0011259767,0.0008664344],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056655225,0.0013756342,0.0028555293,0.001528296,0.0010503147,0.003457049,0.001978563,0.0028154764,0.0039384775],"category_scores_gemma":[0.029143171,0.0013226048,0.0011668107,0.0010802153,0.0059913746,0.005216358,0.0040138317,0.0046826666,0.00020785297],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002815122,0.00010164299,0.0005812245,0.00023133167,0.000080920865,0.00014172477,0.00023947268,0.29326192,0.0016115805,0.69610894,0.0010176792,0.00634207],"study_design_scores_gemma":[0.000047731508,0.00007689364,0.00018150118,0.000030806994,0.000012934031,0.000024115192,0.000039313414,0.7325477,0.00030158152,0.26650152,0.00021770941,0.000018190469],"about_ca_topic_score_codex":0.0037929746,"about_ca_topic_score_gemma":0.0018712521,"teacher_disagreement_score":0.0056655225,"about_ca_system_score_codex":0.002904131,"about_ca_system_score_gemma":0.0029027741,"threshold_uncertainty_score":0.02996254},"labels":[],"label_agreement":null},{"id":"W1799762961","doi":"10.1613/jair.904","title":"Accelerating Reinforcement Learning by Composing Solutions of Automatically Identified Subtasks","year":2002,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Government of Ontario","keywords":"Reinforcement learning; Exploit; Partition (number theory); Function (biology); Base (topology); State space; Reinforcement; Function approximation","score_opus":0.2748869913500262,"score_gpt":0.396747064427609,"score_spread":0.12186007307758279,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1799762961","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09591879,0.00008921867,0.89621115,0.0000960365,0.00004728784,0.00025450962,0.000025407078,0.004415188,0.002942406],"genre_scores_gemma":[0.5686299,0.00009953748,0.42710024,0.00008123036,0.000029740942,0.00031292843,0.000116182266,0.0002664943,0.0033636314],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995541,0.000107465086,0.000032047476,0.00012647721,0.00012871446,0.000051188283],"domain_scores_gemma":[0.99844235,0.00079346384,0.00018890125,0.00028410708,0.00019159587,0.00009956884],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011456525,0.00093502674,0.0007912785,0.00046153084,0.0003441231,0.0005655198,0.0015439946,0.00071445,0.0036160012],"category_scores_gemma":[0.0046338225,0.00046985666,0.00043597078,0.00029795733,0.00060593535,0.0012424581,0.0012177264,0.0011714023,0.0010704844],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005961486,0.000903324,0.00233347,0.00019335368,0.00010751874,0.00018838246,0.00033131673,0.31322524,0.06351679,0.0066761123,0.0016915315,0.6102368],"study_design_scores_gemma":[0.00009314164,0.00027275653,0.0004715872,0.000014517252,0.000036111374,0.00009134388,0.000026896467,0.96743083,0.023522757,0.0055514593,0.0024683003,0.000020313966],"about_ca_topic_score_codex":0.0014857153,"about_ca_topic_score_gemma":0.0012978375,"teacher_disagreement_score":0.0036160012,"about_ca_system_score_codex":0.00044604554,"about_ca_system_score_gemma":0.0006759282,"threshold_uncertainty_score":0.012096763},"labels":[],"label_agreement":null},{"id":"W1801772197","doi":"10.1007/978-3-540-72665-4_5","title":"R-FRTDP: A Real-Time DP Algorithm with Tight Bounds for a Stochastic Resource Allocation Problem","year":2007,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Markov decision process; Mathematical optimization; Resource allocation; Heuristic; Context (archaeology); Task (project management); Stochastic programming; Dynamic programming; Resource (disambiguation); Markov process; Algorithm; Artificial intelligence; Mathematics","score_opus":0.014734208176855099,"score_gpt":0.24645569856153343,"score_spread":0.23172149038467832,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1801772197","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0031073492,0.0002464076,0.9919859,0.00026210875,0.0001297275,0.00007683329,0.00009243344,0.0010997772,0.0029994168],"genre_scores_gemma":[0.11917711,0.00024990295,0.8747909,0.00032693858,0.00012788674,0.00039057338,0.00031310137,0.00052626035,0.0040974],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99859565,0.0004119841,0.00008434485,0.00032688896,0.00034474104,0.00023631423],"domain_scores_gemma":[0.99824846,0.0010899758,0.000106894775,0.00020032012,0.00021708492,0.00013735147],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024691902,0.0015379471,0.0022518332,0.0009067666,0.0006215785,0.0018368555,0.003263149,0.0030512097,0.007910232],"category_scores_gemma":[0.00730774,0.0010809617,0.0011984697,0.001283553,0.001039363,0.0022311956,0.0026981735,0.0036162108,0.0015621508],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035370758,0.00023464518,0.0001946472,0.00025685955,0.00007813994,0.00009352171,0.00007709727,0.7282903,0.002679682,0.04804713,0.014280319,0.20541395],"study_design_scores_gemma":[0.000061250685,0.00002743813,0.000021838563,0.000009276,0.000007663776,0.000022527198,0.000005645323,0.9874578,0.0004062571,0.01057623,0.0013977147,0.000006432041],"about_ca_topic_score_codex":0.0052438015,"about_ca_topic_score_gemma":0.005504174,"teacher_disagreement_score":0.007910232,"about_ca_system_score_codex":0.0016651786,"about_ca_system_score_gemma":0.0032082302,"threshold_uncertainty_score":0.026462317},"labels":[],"label_agreement":null},{"id":"W180335381","doi":"","title":"Modeling Students' Emotions to Improve Learning with Educational Games","year":2001,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Function (biology); State (computer science); Personality; Human–computer interaction; Psychology; Cognitive psychology; Artificial intelligence; Cognitive science; Social psychology","score_opus":0.014679573392387055,"score_gpt":0.2826964279880916,"score_spread":0.2680168545957045,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W180335381","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.54965335,0.00032777377,0.43701017,0.0007057644,0.00005290422,0.00023868655,0.00006054004,0.0002604392,0.01169044],"genre_scores_gemma":[0.97713155,0.000114942224,0.021200262,0.000029479823,0.000006005765,0.00008525016,0.000021528625,0.000011151769,0.0013998316],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99959713,0.00024170497,0.000017560202,0.000046673653,0.000045524594,0.000051479972],"domain_scores_gemma":[0.99884796,0.00077726203,0.0001120057,0.000059891296,0.00011195583,0.000090876754],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009495432,0.00076283235,0.00031479727,0.00025407475,0.00017813452,0.0010088225,0.0007071573,0.00047459634,0.00161118],"category_scores_gemma":[0.0040871673,0.00021544597,0.0003681798,0.00015880223,0.00034213046,0.00097454234,0.00055893575,0.00081865414,0.0001659568],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024356505,0.0006823782,0.011412872,0.00011712577,0.00007556944,0.00007850384,0.0004981746,0.9157065,0.004925278,0.019583808,0.0005626789,0.04611351],"study_design_scores_gemma":[0.00002114175,0.00011786603,0.0013402557,0.0000071312015,0.000020939868,0.0000074592726,0.000038483617,0.9918183,0.0010581204,0.004981941,0.0005822316,0.000006058703],"about_ca_topic_score_codex":0.002206725,"about_ca_topic_score_gemma":0.0038006883,"teacher_disagreement_score":0.002206725,"about_ca_system_score_codex":0.00075070444,"about_ca_system_score_gemma":0.00038113692,"threshold_uncertainty_score":0.0054467916},"labels":[],"label_agreement":null},{"id":"W18175453","doi":"10.1525/tph.2007.29.3.87","title":"Algorithm-Directed Exploration for Model-Based Reinforcement Learning in Factored MDPs","year":2002,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Oracle; Computer science; Planner; Mathematical optimization; Linear programming; Artificial intelligence; Machine learning; Algorithm; Mathematics","score_opus":0.08004596865886063,"score_gpt":0.30279357989068106,"score_spread":0.22274761123182044,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W18175453","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.045031324,0.00046609307,0.9492612,0.0005132784,0.000062387066,0.00012326222,0.000079661666,0.0005550835,0.0039077075],"genre_scores_gemma":[0.89147455,0.0002160717,0.10434739,0.00018697558,0.00004577282,0.00048936304,0.00014594622,0.00010411423,0.0029897182],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99920976,0.00041891736,0.000044916513,0.00010633508,0.000098438264,0.00012168807],"domain_scores_gemma":[0.9931942,0.0057735844,0.00031466145,0.00017514381,0.00032899022,0.00021335408],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031895316,0.0013355146,0.0021679013,0.00078459387,0.0005857314,0.0010293012,0.0015926887,0.0017171492,0.004118219],"category_scores_gemma":[0.009815553,0.0008343204,0.0007095321,0.00057608413,0.0016079741,0.0013642027,0.002034388,0.001959056,0.0003698321],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000057988604,0.000024753768,0.00029754636,0.00004055703,0.000022209204,0.00002620405,0.000036881036,0.9854154,0.00008188817,0.0062962607,0.00030268834,0.007397663],"study_design_scores_gemma":[0.000017925779,0.00001721551,0.000015798443,0.000004730849,0.0000027701797,0.0000026091918,0.0000036547199,0.9958584,0.00003254777,0.0039506,0.00009172709,0.0000020313257],"about_ca_topic_score_codex":0.010568295,"about_ca_topic_score_gemma":0.0088437805,"teacher_disagreement_score":0.010568295,"about_ca_system_score_codex":0.001545161,"about_ca_system_score_gemma":0.0019377006,"threshold_uncertainty_score":0.021013558},"labels":[],"label_agreement":null},{"id":"W1821963137","doi":"","title":"State similarity based approach for improving performance in RL","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Reinforcement learning; Similarity (geometry); Computer science; Context (archaeology); State (computer science); Artificial intelligence; Function (biology); Tree (set theory); Bellman equation; Action (physics); Machine learning; Data mining; Mathematical optimization; Algorithm; Mathematics","score_opus":0.022259558712616864,"score_gpt":0.24901772870665248,"score_spread":0.2267581699940356,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1821963137","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0161222,0.00024434002,0.9804452,0.000096867785,0.000033160162,0.00004911713,0.000019299516,0.0012325367,0.0017573172],"genre_scores_gemma":[0.6607089,0.00024879235,0.33645555,0.0001650861,0.00007801076,0.0002092807,0.000120955,0.00020401398,0.0018094246],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99832803,0.00062179327,0.00011631532,0.00024394075,0.0005880674,0.00010190555],"domain_scores_gemma":[0.9965738,0.0019301424,0.00024320715,0.0005565878,0.0005847477,0.00011159147],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020701282,0.000822932,0.0013205173,0.001192112,0.0005699845,0.0010632494,0.0016517808,0.0013351026,0.00228664],"category_scores_gemma":[0.0070716967,0.0004382256,0.00062053645,0.00088286586,0.0010561771,0.0024895652,0.0016270461,0.001559342,0.0006156112],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002945212,0.00036434576,0.002111668,0.00016577975,0.00012937983,0.0001154332,0.00024639326,0.61997736,0.013423641,0.043912943,0.0015965829,0.31766194],"study_design_scores_gemma":[0.000014477418,0.00013117414,0.00018877332,0.0000055303367,0.0000133072635,0.000030547,0.000007642411,0.9882997,0.0028331853,0.007893566,0.0005708934,0.0000110384335],"about_ca_topic_score_codex":0.001729132,"about_ca_topic_score_gemma":0.0015850615,"teacher_disagreement_score":0.00228664,"about_ca_system_score_codex":0.0009810551,"about_ca_system_score_gemma":0.0010943412,"threshold_uncertainty_score":0.010948002},"labels":[],"label_agreement":null},{"id":"W183139520","doi":"","title":"Stochastic local search for POMDP controllers","year":2004,"lang":"en","type":"article","venue":"TSpace","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Partially observable Markov decision process; Heuristics; Computer science; Markov decision process; Mathematical optimization; Dynamic programming; Observable; Controller (irrigation); State (computer science); Markov process; Markov chain; Mathematics; Algorithm; Machine learning; Markov model","score_opus":0.02851222499342767,"score_gpt":0.32275675840771967,"score_spread":0.294244533414292,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W183139520","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013094518,0.00027261794,0.9831315,0.00014573897,0.000021805845,0.00004628544,0.000033976954,0.00047786016,0.0027757164],"genre_scores_gemma":[0.7484934,0.0003308916,0.24704036,0.00015005031,0.000041890195,0.00046764035,0.00013679356,0.00015555815,0.003183521],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99940276,0.0002508289,0.00002859216,0.00010744103,0.00015621014,0.000054205422],"domain_scores_gemma":[0.99769235,0.0018013734,0.00016883593,0.00008516412,0.00015949432,0.000092765],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015290952,0.0008244778,0.0012390498,0.0006956168,0.0004982582,0.00085949944,0.00097296474,0.00093121355,0.003243274],"category_scores_gemma":[0.004651055,0.00058648695,0.0005787627,0.0005439913,0.0014598602,0.00095908146,0.0011729851,0.0012687021,0.00042072817],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000027519409,0.000016186452,0.000119086224,0.00005871574,0.000017369837,0.000026189628,0.000026485812,0.97567683,0.00038672204,0.013221234,0.00039823173,0.010025428],"study_design_scores_gemma":[0.000012302669,0.000012945216,0.000014999224,0.000004546548,0.0000025594704,0.0000033695878,0.0000042642246,0.9943032,0.00013842419,0.0053050183,0.0001963281,0.0000020614052],"about_ca_topic_score_codex":0.0032619499,"about_ca_topic_score_gemma":0.0029560865,"teacher_disagreement_score":0.0032619499,"about_ca_system_score_codex":0.001216993,"about_ca_system_score_gemma":0.0014305837,"threshold_uncertainty_score":0.0108498335},"labels":[],"label_agreement":null},{"id":"W185706336","doi":"","title":"Labeled Initialized Adaptive Play Q-learning for Stochastic Games","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Q-learning; Computer science; Context (archaeology); Process (computing); Artificial intelligence; Optimal stopping; Point (geometry); Mathematical optimization; Adaptive learning; Grid; Quality (philosophy); Mathematics; Reinforcement learning","score_opus":0.027898508620810496,"score_gpt":0.2864012359353325,"score_spread":0.258502727314522,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W185706336","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004565822,0.00008814366,0.99433374,0.000098435405,0.000013711657,0.000050214694,0.000009064719,0.000079640275,0.0007612998],"genre_scores_gemma":[0.5434899,0.00031777105,0.4509,0.0002681192,0.00006610962,0.0007494056,0.00011156068,0.00011278641,0.0039844876],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99770015,0.0014440361,0.000088921435,0.00028994534,0.00032485995,0.00015203578],"domain_scores_gemma":[0.99154496,0.0065177623,0.000546781,0.0003520384,0.00074552896,0.0002929186],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044136867,0.0011891604,0.001172053,0.0006901244,0.00075150543,0.0012947854,0.0020441397,0.0012936353,0.0028567242],"category_scores_gemma":[0.015656902,0.0005361486,0.00052357453,0.0005786441,0.0028345734,0.0019177125,0.0015870458,0.002444875,0.0004382448],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016689514,0.0001050128,0.0011189904,0.00014038067,0.000048268972,0.00010931335,0.0002165503,0.7336448,0.001264497,0.21801013,0.0014023182,0.04377291],"study_design_scores_gemma":[0.000016243543,0.00003429146,0.000042237265,0.000007509867,0.000003272721,0.000007691728,0.000005610642,0.96954554,0.0002579391,0.02973291,0.00034143295,0.000005281408],"about_ca_topic_score_codex":0.003549172,"about_ca_topic_score_gemma":0.0025907587,"teacher_disagreement_score":0.0044136867,"about_ca_system_score_codex":0.0020462165,"about_ca_system_score_gemma":0.001855391,"threshold_uncertainty_score":0.023342073},"labels":[],"label_agreement":null},{"id":"W1862757251","doi":"10.1007/978-3-642-27645-3_11","title":"Bayesian Reinforcement Learning","year":2012,"lang":"en","type":"book-chapter","venue":"Adaptation, learning, and optimization","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":64,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Bayesian probability; Artificial intelligence; Computer science; Machine learning; Posterior probability; Prior probability; Domain (mathematical analysis); Bayesian inference; Function (biology); Bellman equation; Mathematics; Mathematical optimization","score_opus":0.015541756629481101,"score_gpt":0.22207336376070658,"score_spread":0.20653160713122548,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1862757251","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0019639258,0.013286888,0.7720085,0.0019027161,0.00069790153,0.00006993876,0.00022545933,0.0009358386,0.20890883],"genre_scores_gemma":[0.25973523,0.027607098,0.35102594,0.0014405756,0.0013116216,0.00046567345,0.0011147953,0.0006878257,0.35661125],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99959284,0.00009560728,0.000015119481,0.000087156346,0.00018203439,0.000027183678],"domain_scores_gemma":[0.9996562,0.00016849636,0.000022136594,0.000053790256,0.00007422722,0.00002511672],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00060149614,0.0009676237,0.00075936835,0.00052691635,0.00036182057,0.0014785243,0.001213073,0.0011584719,0.01932428],"category_scores_gemma":[0.0020475315,0.0004153423,0.00039971157,0.0006791801,0.001028034,0.0013483905,0.00096196367,0.0019262212,0.006463266],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000041471012,0.00008202713,0.00017824718,0.00024152023,0.000042602176,0.000041481646,0.00006927221,0.044967443,0.0012935236,0.47111237,0.05185455,0.43007553],"study_design_scores_gemma":[0.000022649405,0.00004149238,0.00033656735,0.000179806,0.000023968327,0.00014473303,0.000025716196,0.14053224,0.0015643262,0.65937525,0.19771065,0.000042652435],"about_ca_topic_score_codex":0.0017204897,"about_ca_topic_score_gemma":0.0025741116,"teacher_disagreement_score":0.01932428,"about_ca_system_score_codex":0.0010596021,"about_ca_system_score_gemma":0.0008514033,"threshold_uncertainty_score":0.064646125},"labels":[],"label_agreement":null},{"id":"W1870822514","doi":"","title":"A Deeper Look at Planning as Learning from Replay","year":2015,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Bellman equation; Markov decision process; Function (biology); Temporal difference learning; Markov chain; Equivalence (formal languages); Artificial intelligence; Markov process; Machine learning; Mathematical optimization; Mathematics","score_opus":0.07534542055669481,"score_gpt":0.32282196044179445,"score_spread":0.24747653988509966,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1870822514","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002378328,0.0016856887,0.99086577,0.0015782308,0.00011820729,0.000021774214,0.000029873861,0.0002397905,0.003082335],"genre_scores_gemma":[0.38590112,0.006757573,0.5892979,0.0019459666,0.0006739963,0.0003111145,0.00017935455,0.0005136695,0.014419305],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988601,0.0004950304,0.00005330454,0.00027927555,0.00024431248,0.00006786358],"domain_scores_gemma":[0.99755764,0.0016392989,0.00018369143,0.00034077134,0.0001723983,0.0001061629],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001978775,0.0011592944,0.0012972337,0.000534215,0.0005485629,0.0022758455,0.0019241432,0.0019769093,0.0052306103],"category_scores_gemma":[0.0069022304,0.0007499401,0.0014929348,0.00075111416,0.0032938153,0.0076393895,0.0020569486,0.005846545,0.00063278334],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009092649,0.000061337756,0.0004556718,0.00034802206,0.00009722259,0.00008865405,0.0004295391,0.2832438,0.0019124337,0.6599524,0.0021147777,0.051205184],"study_design_scores_gemma":[0.00002488836,0.00012313016,0.00014311941,0.00011300114,0.000026158585,0.00006488646,0.00006658515,0.5853122,0.0010841035,0.4031106,0.0098879095,0.00004341671],"about_ca_topic_score_codex":0.00485964,"about_ca_topic_score_gemma":0.002755284,"teacher_disagreement_score":0.0052306103,"about_ca_system_score_codex":0.0017181558,"about_ca_system_score_gemma":0.0013174,"threshold_uncertainty_score":0.017498136},"labels":[],"label_agreement":null},{"id":"W187740018","doi":"10.1007/978-1-4757-3561-1_5","title":"On Optimal Policies of Multichain Finite State Compact Action Markov Decision Processes","year":2002,"lang":"en","type":"book-chapter","venue":"Advances in computational management science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Group for Research in Decision Analysis; HEC Montréal","funders":"","keywords":"Markov decision process; Markov process; Action (physics); Set (abstract data type); Mathematical optimization; Markov chain; State (computer science); Finite set; Markov model; Computer science; Mathematics; Mathematical economics; Applied mathematics; Algorithm","score_opus":0.02400273753884332,"score_gpt":0.30644236955273346,"score_spread":0.28243963201389016,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W187740018","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04324519,0.002021669,0.939275,0.0007654661,0.00013716926,0.00006486428,0.00018631354,0.00017737145,0.014126958],"genre_scores_gemma":[0.75775415,0.0037735293,0.22810046,0.00029953278,0.00021683541,0.00031041095,0.00046156687,0.00021837316,0.008865212],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9990152,0.00044736467,0.00005579037,0.0001593039,0.00019419064,0.00012807535],"domain_scores_gemma":[0.99540097,0.0037678261,0.0002704766,0.00017426348,0.00020592361,0.00018062306],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023416358,0.0010862558,0.0017431107,0.00079128344,0.00074342085,0.0018156206,0.0017343765,0.0013116772,0.004963823],"category_scores_gemma":[0.007912301,0.00093812356,0.00082924817,0.0014227878,0.00249824,0.0031851318,0.0018833822,0.00256826,0.00032388358],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007655281,0.00004560752,0.00026189937,0.000101648286,0.000033797707,0.000045131754,0.00012901673,0.5483039,0.00052769855,0.43212104,0.0012414121,0.017112311],"study_design_scores_gemma":[0.0000141243045,0.000017184122,0.00007816427,0.000022330949,0.0000054392954,0.000008550474,0.000013009871,0.6248916,0.0001293514,0.37430632,0.00050574786,0.000008311386],"about_ca_topic_score_codex":0.0062053637,"about_ca_topic_score_gemma":0.0055935336,"teacher_disagreement_score":0.0062053637,"about_ca_system_score_codex":0.0026797964,"about_ca_system_score_gemma":0.0019216449,"threshold_uncertainty_score":0.019443393},"labels":[],"label_agreement":null},{"id":"W1888020014","doi":"10.1109/icsmc.1999.825317","title":"An approach to design autonomous agents within ModSAF","year":2003,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Royal Military College of Canada","funders":"","keywords":"Exploit; Robustness (evolution); Computer science; Human–computer interaction; Autonomous agent; Artificial intelligence; Distributed computing; Computer security","score_opus":0.05927460155812025,"score_gpt":0.27625970169207037,"score_spread":0.21698510013395012,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1888020014","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007098188,0.00008932408,0.98675704,0.0001953801,0.000045798522,0.000101553924,0.000015332918,0.00048377845,0.0052136453],"genre_scores_gemma":[0.111610524,0.00011001267,0.8833279,0.00014874284,0.000016561724,0.0002826716,0.000049130187,0.00009184839,0.0043626367],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997043,0.00007054864,0.000016717919,0.000051440486,0.00011335848,0.00004366809],"domain_scores_gemma":[0.99978644,0.000048917234,0.000029538267,0.000049608025,0.000056414974,0.000029099878],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00077464286,0.00046823366,0.00036578684,0.0003622805,0.0007534799,0.00080250675,0.0013800892,0.00091608806,0.0025089062],"category_scores_gemma":[0.0010318146,0.00032210222,0.0006343479,0.00020445212,0.0008080647,0.00075741095,0.0013727681,0.0009315858,0.00074856065],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013960233,0.00018656702,0.002373255,0.00031826668,0.00013959447,0.00028125747,0.00075456494,0.4084851,0.019872898,0.26597038,0.005265575,0.29621288],"study_design_scores_gemma":[0.00007665002,0.00025174965,0.00021045686,0.000060033188,0.000049454346,0.00026242435,0.00012526962,0.8417156,0.011753738,0.07113415,0.07433256,0.000028006933],"about_ca_topic_score_codex":0.0014476699,"about_ca_topic_score_gemma":0.0021188254,"teacher_disagreement_score":0.0025089062,"about_ca_system_score_codex":0.0004477572,"about_ca_system_score_gemma":0.0011463189,"threshold_uncertainty_score":0.008393109},"labels":[],"label_agreement":null},{"id":"W189498254","doi":"10.1609/icaps.v19i1.13365","title":"Minimal Sufficient Explanations for Factored Markov Decision Processes","year":2009,"lang":"en","type":"article","venue":"Proceedings of the International Conference on Automated Planning and Scheduling","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":75,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Task (project management); Markov decision process; Computer science; Set (abstract data type); Template; Domain (mathematical analysis); Probabilistic logic; Selection (genetic algorithm); Markov chain; Markov process; Markov model; Machine learning; Artificial intelligence; Programming language; Mathematics; Engineering","score_opus":0.04320861166281764,"score_gpt":0.3100428253267046,"score_spread":0.26683421366388693,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W189498254","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023713725,0.00019038138,0.97004557,0.00085674564,0.000040656912,0.0002357099,0.0011784596,0.0013417602,0.002396938],"genre_scores_gemma":[0.42088416,0.00029869768,0.57330936,0.00030822377,0.000073246505,0.0008751899,0.002628358,0.00025960934,0.0013631706],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99534506,0.0018267543,0.0005656442,0.00076219183,0.0011518163,0.00034855574],"domain_scores_gemma":[0.96893054,0.025795193,0.0016157704,0.0015908605,0.001607188,0.0004604401],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039563263,0.0014786744,0.0010229776,0.0016392781,0.00096122664,0.0017447792,0.001647864,0.0022222372,0.00713283],"category_scores_gemma":[0.03379838,0.001092509,0.0022322438,0.00076054304,0.0017774819,0.003109493,0.0025203342,0.002817871,0.000639118],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037960956,0.0002029835,0.0033968745,0.00097584364,0.00020239646,0.0012753705,0.001447881,0.3859012,0.0036900023,0.528806,0.0065621627,0.06715961],"study_design_scores_gemma":[0.00013231946,0.00007307995,0.00032031225,0.00013780636,0.000058347883,0.00010953382,0.00010747447,0.5657699,0.0024153832,0.42630965,0.004524836,0.00004139426],"about_ca_topic_score_codex":0.0029057472,"about_ca_topic_score_gemma":0.0053374837,"teacher_disagreement_score":0.00713283,"about_ca_system_score_codex":0.001609738,"about_ca_system_score_gemma":0.0026780993,"threshold_uncertainty_score":0.023861706},"labels":[],"label_agreement":null},{"id":"W189510620","doi":"10.5555/1838206.1838300","title":"Optimal policy switching algorithms for reinforcement learning","year":2010,"lang":"en","type":"article","venue":"Adaptive Agents and Multi-Agents Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Task (project management); Function approximation; Term (time); Function (biology); Artificial intelligence; Mathematical optimization; Q-learning; Machine learning; Algorithm; Mathematics; Artificial neural network; Engineering","score_opus":0.05548713756434804,"score_gpt":0.3163627497273219,"score_spread":0.2608756121629739,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W189510620","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005134111,0.0004003887,0.991503,0.00016927069,0.00004741728,0.0000595657,0.000026472608,0.0003059002,0.0023538356],"genre_scores_gemma":[0.59698373,0.0008321248,0.3954748,0.0003412707,0.00013064551,0.0008130967,0.00021436438,0.00019879348,0.005011176],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990702,0.0004071347,0.000053261072,0.00015116773,0.00022086462,0.00009743308],"domain_scores_gemma":[0.99728847,0.002113807,0.00016549052,0.00010947858,0.00022644803,0.00009628199],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020821374,0.001189238,0.0014650548,0.00079528,0.00044194478,0.0010300056,0.0015716777,0.0014595087,0.0046946364],"category_scores_gemma":[0.007191912,0.0005578952,0.0005727783,0.0007114283,0.0014098543,0.0013264138,0.0013958771,0.0025074554,0.0006836968],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008640675,0.0000929384,0.0003283193,0.000088297624,0.000041411033,0.000031133077,0.00006943924,0.86359227,0.0004243212,0.06594282,0.0016092189,0.067693375],"study_design_scores_gemma":[0.0000204624,0.00001671238,0.000021932168,0.0000069014277,0.000003713704,0.000004511725,0.0000035021596,0.9754724,0.00011347684,0.023947462,0.00038518323,0.0000036394513],"about_ca_topic_score_codex":0.003093485,"about_ca_topic_score_gemma":0.00223926,"teacher_disagreement_score":0.0046946364,"about_ca_system_score_codex":0.001451451,"about_ca_system_score_gemma":0.0014149252,"threshold_uncertainty_score":0.015705109},"labels":[],"label_agreement":null},{"id":"W1895872655","doi":"10.1002/9780470400531.eorms0714","title":"Reinforcement Learning Algorithms for<scp>MDPs</scp>","year":2011,"lang":"en","type":"other","venue":"Wiley Encyclopedia of Operations Research and Management Science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Reinforcement; Artificial intelligence; Learning classifier system; Machine learning; Focus (optics); Engineering","score_opus":0.04620281988043152,"score_gpt":0.32626495270884837,"score_spread":0.28006213282841685,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1895872655","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016259044,0.00048618205,0.9750117,0.0005794013,0.00005840645,0.00008489,0.00005722141,0.00033048322,0.007132602],"genre_scores_gemma":[0.7277766,0.0005772987,0.26380557,0.00023442038,0.000115671515,0.00043559112,0.00018841383,0.0001611512,0.006705333],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99947065,0.00021676633,0.000029511004,0.00007538997,0.00014371553,0.0000640253],"domain_scores_gemma":[0.99762934,0.0016482379,0.00019285968,0.00011556377,0.0003000531,0.00011399064],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015210037,0.00073102117,0.0010345835,0.00057194306,0.0005152264,0.0009725865,0.0013368387,0.0010649672,0.0042815045],"category_scores_gemma":[0.0052621,0.00034558366,0.0004105322,0.0006756886,0.0012081394,0.00090014987,0.0015388319,0.001795173,0.0005147647],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003164847,0.0000303936,0.00019524379,0.00003848558,0.000015986729,0.000031685828,0.00002607161,0.9344907,0.00015343199,0.037614025,0.0014642586,0.025907997],"study_design_scores_gemma":[0.000011795621,0.000007608022,0.00001563778,0.000004309196,0.0000013810647,0.000004489816,0.0000021850442,0.9853324,0.00006465807,0.01422846,0.0003253394,0.0000016983822],"about_ca_topic_score_codex":0.006266804,"about_ca_topic_score_gemma":0.00394158,"teacher_disagreement_score":0.006266804,"about_ca_system_score_codex":0.0014259482,"about_ca_system_score_gemma":0.0012984358,"threshold_uncertainty_score":0.014322996},"labels":[],"label_agreement":null},{"id":"W1898595911","doi":"10.1109/roman.1995.531970","title":"Elements of artificial emotion","year":2002,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Cisco Systems (Canada)","funders":"","keywords":"Implementation; Situated; Curiosity; Computer science; Affection; Anger; Action selection; Embodied cognition; Artificial intelligence; Action (physics); Human–computer interaction; Robot; Autonomous agent; Cognitive science; Cognitive psychology; Psychology; Social psychology; Software engineering","score_opus":0.037571531359785026,"score_gpt":0.24743231552303907,"score_spread":0.20986078416325404,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1898595911","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028751595,0.005588701,0.36090592,0.017904526,0.0011153757,0.00018906439,0.0002933684,0.0005188159,0.58473265],"genre_scores_gemma":[0.7243807,0.0033927634,0.20237681,0.0025564507,0.0005091074,0.00072398555,0.0004038629,0.00019966903,0.06545662],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9993954,0.00021211784,0.000044287368,0.00015420235,0.00014493501,0.00004899169],"domain_scores_gemma":[0.9996183,0.0001294615,0.000035917918,0.00009429274,0.000065823675,0.000056292098],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006922985,0.0004662537,0.00029943275,0.00055186683,0.0008511568,0.0031425671,0.0009072453,0.0010012472,0.0077668023],"category_scores_gemma":[0.0017097627,0.00026728603,0.000441898,0.00037092296,0.0057022753,0.0031564825,0.0022854926,0.0019325127,0.001396413],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000020779886,0.00001535498,0.00036104626,0.00009052035,0.0000104275005,0.00006680996,0.0011337268,0.0013792738,0.0018320165,0.9719173,0.0018846424,0.021288099],"study_design_scores_gemma":[0.000014802435,0.000033797103,0.0005688603,0.00007192783,0.000013985305,0.0001775426,0.00051481795,0.005486568,0.0008831486,0.8655997,0.12661874,0.000016050877],"about_ca_topic_score_codex":0.0004906069,"about_ca_topic_score_gemma":0.0004110846,"teacher_disagreement_score":0.0077668023,"about_ca_system_score_codex":0.000994794,"about_ca_system_score_gemma":0.0004302738,"threshold_uncertainty_score":0.025982559},"labels":[],"label_agreement":null},{"id":"W190427941","doi":"10.20965/jaciii.2006.p0578","title":"Opposition-Based Reinforcement Learning","year":2006,"lang":"en","type":"article","venue":"Journal of Advanced Computational Intelligence and Intelligent Informatics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":207,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Expediting; Artificial intelligence; A priori and a posteriori; Probabilistic logic; Grid; Machine learning; Opposition (politics); Convergence (economics); Learning classifier system; Engineering; Mathematics","score_opus":0.01614136464924268,"score_gpt":0.26280084582795943,"score_spread":0.24665948117871675,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W190427941","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013721921,0.00027287813,0.98011655,0.00016298804,0.00006880312,0.00009176248,0.00002580292,0.00031309988,0.005226076],"genre_scores_gemma":[0.8651396,0.00032778495,0.12949096,0.00024592163,0.000053942582,0.00036171693,0.00008603418,0.000047892772,0.004246194],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992041,0.00033493165,0.000041086787,0.00011160924,0.00022511417,0.00008315192],"domain_scores_gemma":[0.99832493,0.0010594331,0.00017028165,0.000101226244,0.00025360595,0.00009047782],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013050422,0.00075727387,0.0013753604,0.00045104936,0.0003827406,0.00075289205,0.0013128552,0.0009079622,0.002439095],"category_scores_gemma":[0.004146397,0.0002634005,0.00047541226,0.000427672,0.0011899394,0.00070895924,0.0010782845,0.0011542558,0.0003888851],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016038716,0.00014951153,0.0009686153,0.00015323966,0.000093112765,0.00015957808,0.00009601682,0.86309284,0.0022274167,0.03708625,0.001675329,0.09413773],"study_design_scores_gemma":[0.000042065363,0.00008306065,0.000084672625,0.000009052339,0.000009039812,0.00003193262,0.0000062944932,0.9886839,0.00047220345,0.009446564,0.0011230467,0.000008128266],"about_ca_topic_score_codex":0.0021048817,"about_ca_topic_score_gemma":0.0015699944,"teacher_disagreement_score":0.002439095,"about_ca_system_score_codex":0.0006690928,"about_ca_system_score_gemma":0.00077618903,"threshold_uncertainty_score":0.008159518},"labels":[],"label_agreement":null},{"id":"W1904406446","doi":"","title":"Using predictive representations to improve generalization in reinforcement learning","year":2005,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Generalization; Reinforcement learning; Computer science; Representation (politics); Artificial intelligence; Task (project management); State (computer science); Machine learning; Grid; Reinforcement; Algorithm; Mathematics; Psychology; Social psychology; Engineering","score_opus":0.029941249022495058,"score_gpt":0.30813073930994245,"score_spread":0.2781894902874474,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1904406446","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08824538,0.00053411,0.9046426,0.0007173379,0.000087128305,0.00006676626,0.000105688596,0.001903695,0.0036972892],"genre_scores_gemma":[0.9241363,0.00030444347,0.0735293,0.00021223906,0.00005621249,0.00011538412,0.00013705548,0.00008319116,0.001425826],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99935144,0.0002627637,0.000043753207,0.00014187102,0.00013282048,0.00006735728],"domain_scores_gemma":[0.9944088,0.0037877634,0.0003545029,0.0009051923,0.00042377994,0.000119871496],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020066756,0.0007429364,0.00083668844,0.00042081182,0.0003070886,0.0008471475,0.0013960294,0.0010580074,0.0026822225],"category_scores_gemma":[0.013565825,0.000397653,0.00043382915,0.00055167725,0.0011865688,0.0040022032,0.0017653686,0.0020882564,0.00040230723],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002709912,0.00017883364,0.0012395888,0.00013689816,0.00006549611,0.00008482992,0.00021279715,0.74499357,0.0036941047,0.040538426,0.0018588052,0.20672579],"study_design_scores_gemma":[0.000027943888,0.00010212972,0.00017350118,0.000012205814,0.00001189785,0.0000170255,0.000011648378,0.965212,0.0012232937,0.032808047,0.00038972448,0.000010550435],"about_ca_topic_score_codex":0.0023305668,"about_ca_topic_score_gemma":0.0018172908,"teacher_disagreement_score":0.0026822225,"about_ca_system_score_codex":0.00074437435,"about_ca_system_score_gemma":0.0007303328,"threshold_uncertainty_score":0.010612488},"labels":[],"label_agreement":null},{"id":"W19532835","doi":"10.1609/aiide.v4i1.18667","title":"Agent Learning Using Action-Dependent Learning Rates in Computer Role-Playing Games","year":2008,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Action (physics); Computer science; Reinforcement learning; Scripting language; Artificial intelligence; Error-driven learning; Learning effect; Human–computer interaction","score_opus":0.07247847678986213,"score_gpt":0.30101875507031534,"score_spread":0.2285402782804532,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W19532835","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049223065,0.00021807976,0.94705296,0.0001884539,0.000044922006,0.000084943626,0.000017182754,0.0008952029,0.0022751582],"genre_scores_gemma":[0.72995305,0.00023707448,0.2677628,0.00010159368,0.000039773466,0.00022237554,0.0000384142,0.0001642427,0.0014806824],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99834204,0.00084321294,0.00012866386,0.00020099012,0.00037626002,0.00010873836],"domain_scores_gemma":[0.99194056,0.0056145233,0.00073358737,0.0005137295,0.0008545994,0.00034303433],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00502616,0.0006692759,0.00073874596,0.00063184655,0.00033587008,0.0011183822,0.0016291699,0.00089663255,0.0010537916],"category_scores_gemma":[0.019907756,0.0005590927,0.00047575723,0.00037090725,0.0010533299,0.002222233,0.0010060995,0.0017475362,0.000300038],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000248859,0.00034207397,0.0034833213,0.00009957858,0.000063209016,0.000095087715,0.0002667681,0.8042923,0.0065558013,0.033993363,0.0007879197,0.14977178],"study_design_scores_gemma":[0.000026354039,0.00006151386,0.00016386154,0.0000068261315,0.000007918889,0.000020864069,0.0000073880196,0.9889683,0.0020887773,0.0082444595,0.000393065,0.000010611875],"about_ca_topic_score_codex":0.00202285,"about_ca_topic_score_gemma":0.00123317,"teacher_disagreement_score":0.00502616,"about_ca_system_score_codex":0.0009822891,"about_ca_system_score_gemma":0.0008801088,"threshold_uncertainty_score":0.026581228},"labels":[],"label_agreement":null},{"id":"W1973749650","doi":"10.1016/j.artint.2012.04.006","title":"Reinforcement learning with limited reinforcement: Using Bayes risk for active learning in POMDPs","year":2012,"lang":"en","type":"article","venue":"Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Partially observable Markov decision process; Reinforcement learning; Computer science; Markov decision process; Task (project management); Artificial intelligence; Machine learning; Action (physics); Variety (cybernetics); Bayes' theorem; Bayesian probability; Markov chain; Markov process; Markov model; Mathematics","score_opus":0.06131471456678903,"score_gpt":0.3031917075469479,"score_spread":0.24187699298015888,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1973749650","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012236627,0.00023327443,0.98628545,0.00017441847,0.00003791055,0.000036225803,0.000020934202,0.00016954188,0.00080559624],"genre_scores_gemma":[0.83399665,0.00026110592,0.16297822,0.00016945937,0.0000976673,0.00033140837,0.00008822446,0.00012231791,0.0019549602],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979107,0.001103418,0.00013090641,0.00028298286,0.00040844776,0.0001636363],"domain_scores_gemma":[0.98716766,0.01080015,0.00048326253,0.00041556882,0.0007593548,0.00037400375],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006558639,0.0013134837,0.0028819386,0.0007208174,0.00063205074,0.0017171835,0.0027137476,0.0021181167,0.0028522385],"category_scores_gemma":[0.019932048,0.0013260456,0.0010163495,0.0005020507,0.0020773534,0.003857485,0.0029193026,0.0031424717,0.0003358923],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016586922,0.00007727094,0.0005300401,0.00010367899,0.000059451027,0.000040185892,0.000064542335,0.9522944,0.000323611,0.019198097,0.00039365338,0.026749123],"study_design_scores_gemma":[0.000010949466,0.0000128379725,0.0000151084905,0.000005709424,0.000004304819,0.000002302532,0.0000017349504,0.9937317,0.00007244762,0.006100242,0.000040147086,0.0000026303387],"about_ca_topic_score_codex":0.0049676728,"about_ca_topic_score_gemma":0.003661803,"teacher_disagreement_score":0.006558639,"about_ca_system_score_codex":0.0013182767,"about_ca_system_score_gemma":0.0017419817,"threshold_uncertainty_score":0.03468585},"labels":[],"label_agreement":null},{"id":"W1978322215","doi":"10.1109/tcyb.2014.2375817","title":"Energy Efficient Execution of POMDP Policies","year":2015,"lang":"en","type":"article","venue":"IEEE Transactions on Cybernetics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Ontario Ministry of Research and Innovation; Natural Sciences and Engineering Research Council of Canada; Ontario Ministry of Research, Innovation and Science; Toronto Rehabilitation Institute; Alzheimer's Association","keywords":"Partially observable Markov decision process; Computer science; Benchmark (surveying); Markov decision process; Compiler; Mathematical optimization; Markov chain; Markov process; Machine learning; Markov model","score_opus":0.027320007767192143,"score_gpt":0.24979685337084656,"score_spread":0.2224768456036544,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1978322215","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12329021,0.00020372831,0.8660585,0.00022110598,0.00004491488,0.00015320341,0.00022521673,0.0030551862,0.006748009],"genre_scores_gemma":[0.8329592,0.00015673862,0.16503558,0.00006223968,0.000011369374,0.00018265001,0.00027806693,0.00018713002,0.0011271401],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999476,0.00013112252,0.000034994442,0.00010335793,0.00017700976,0.000077440454],"domain_scores_gemma":[0.9985092,0.0009804282,0.00009791793,0.00022828608,0.000141217,0.000042849893],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006555704,0.0007091759,0.00060869404,0.00036210698,0.000365014,0.00069815683,0.00070086267,0.0004868314,0.0024243193],"category_scores_gemma":[0.003260377,0.00039856439,0.00054140855,0.00029471735,0.0007668527,0.0008227658,0.0006516513,0.0009876253,0.00031175715],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000711166,0.00003788067,0.00043796893,0.000049619037,0.000013267406,0.00003614433,0.00004696456,0.9723521,0.0023001065,0.003794387,0.00025053465,0.020610007],"study_design_scores_gemma":[0.000014609928,0.000031098454,0.00011029482,0.0000062981103,0.000006421662,0.000008058534,0.000016961832,0.99354917,0.0022638573,0.003529672,0.00046017338,0.0000034661964],"about_ca_topic_score_codex":0.005112696,"about_ca_topic_score_gemma":0.0050668227,"teacher_disagreement_score":0.005112696,"about_ca_system_score_codex":0.0008167233,"about_ca_system_score_gemma":0.0016274366,"threshold_uncertainty_score":0.01016587},"labels":[],"label_agreement":null},{"id":"W1978640853","doi":"10.1533/abbi.2005.0021","title":"A Game Theoretic Approach to Swarm Robotics","year":2006,"lang":"en","type":"article","venue":"Applied Bionics and Biomechanics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Swarm robotics; Robotics; Artificial intelligence; Swarm behaviour; Computer science; Human–computer interaction; Engineering; Robot","score_opus":0.008186755589107645,"score_gpt":0.19745562999519717,"score_spread":0.18926887440608953,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1978640853","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0028727606,0.0023464218,0.95483863,0.0026536663,0.00028378482,0.000095372816,0.00004370735,0.00006477677,0.036800824],"genre_scores_gemma":[0.5206858,0.008477289,0.4395662,0.0016688951,0.0011300294,0.0009987026,0.00012304552,0.00008680084,0.02726328],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99914443,0.0004683996,0.000038184422,0.00008330363,0.00020797359,0.00005760916],"domain_scores_gemma":[0.99943656,0.00035220393,0.000045265602,0.00004213034,0.00005989277,0.000063907275],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00096924347,0.0013344187,0.00095638545,0.00076975673,0.0009719469,0.001867853,0.001708857,0.001799657,0.0037396722],"category_scores_gemma":[0.0017581659,0.00038776858,0.0011396935,0.00074777537,0.0036684303,0.0020097033,0.0014931699,0.002877289,0.0006005052],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000056437216,0.000019727702,0.00005671257,0.00007269736,0.000022258855,0.00006904662,0.0001110409,0.044064827,0.0003710892,0.9485143,0.0010745095,0.0056181606],"study_design_scores_gemma":[0.000017689252,0.000043054733,0.000045639084,0.00003077931,0.000010495888,0.000057431356,0.000058658556,0.1458729,0.0001411219,0.8407726,0.01293631,0.000013161517],"about_ca_topic_score_codex":0.0020963985,"about_ca_topic_score_gemma":0.001666242,"teacher_disagreement_score":0.0037396722,"about_ca_system_score_codex":0.0016015886,"about_ca_system_score_gemma":0.0011829614,"threshold_uncertainty_score":0.0125104785},"labels":[],"label_agreement":null},{"id":"W1979239334","doi":"10.1109/acc.2010.5530771","title":"An investigation of guarding a territory problem in a grid world","year":2010,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Reinforcement learning; Computer science; Minimax; Grid; Artificial intelligence; Climbing; Hill climbing; Machine learning; Mathematical optimization; Geography; Mathematics","score_opus":0.011641736698358983,"score_gpt":0.24617533696483948,"score_spread":0.23453360026648049,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1979239334","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13079774,0.00093101757,0.8501005,0.0008677054,0.00010243552,0.00010300911,0.000088687426,0.00013699612,0.016871918],"genre_scores_gemma":[0.9284443,0.0007538447,0.06602409,0.0000904375,0.000041580286,0.000090480804,0.00006906089,0.00003346395,0.004452664],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996456,0.00014995124,0.000013279364,0.00006638371,0.00005972418,0.000065095825],"domain_scores_gemma":[0.99922013,0.00045799985,0.00011149932,0.000046175068,0.00006280959,0.00010135779],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00041078197,0.00047857797,0.0008164871,0.00029759735,0.00083143474,0.0008651047,0.0010838938,0.0011473752,0.0025378054],"category_scores_gemma":[0.0016367268,0.0002695462,0.00059445243,0.00044954979,0.0012401426,0.002197482,0.0012806698,0.0010465402,0.00019281509],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009792346,0.00006436656,0.00227103,0.0001888374,0.00006324651,0.0008588509,0.00017658746,0.83899015,0.0025076906,0.13769016,0.0014575192,0.015633596],"study_design_scores_gemma":[0.000018667834,0.000065778295,0.000323227,0.000010011209,0.000014925752,0.00015408544,0.00010412459,0.9698948,0.00031741447,0.027563684,0.0015229573,0.000010170037],"about_ca_topic_score_codex":0.005486682,"about_ca_topic_score_gemma":0.003330489,"teacher_disagreement_score":0.005486682,"about_ca_system_score_codex":0.0006577069,"about_ca_system_score_gemma":0.0007140267,"threshold_uncertainty_score":0.010909438},"labels":[],"label_agreement":null},{"id":"W1980722163","doi":"10.1007/s10846-015-0222-2","title":"Multiple Model Q-Learning for Stochastic Asynchronous Rewards","year":2015,"lang":"en","type":"article","venue":"Journal of Intelligent & Robotic Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Royal Military College of Canada; Carleton University","funders":"","keywords":"Reinforcement learning; Asynchronous communication; Mobile robot; Computer science; Convergence (economics); Stochastic approximation; SIGNAL (programming language); Robot; Relaxation (psychology); Simulation; Real-time computing; Artificial intelligence","score_opus":0.060132220187145424,"score_gpt":0.28683430531278326,"score_spread":0.22670208512563783,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1980722163","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013825069,0.00036483567,0.9830562,0.0005298132,0.000060981612,0.000080118465,0.000071655864,0.0002114939,0.0017999188],"genre_scores_gemma":[0.825991,0.00041543547,0.16006128,0.00036798033,0.0001551225,0.000563345,0.00025972555,0.00018523146,0.012000961],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976926,0.0011372331,0.000115846975,0.00043578743,0.0003080007,0.00031053007],"domain_scores_gemma":[0.98051375,0.016735358,0.00077044853,0.00050852896,0.0009656633,0.00050628465],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006669304,0.0013599448,0.0037131594,0.0009804243,0.00095037447,0.0020511984,0.0039615156,0.0031769474,0.008291143],"category_scores_gemma":[0.021456394,0.0014281309,0.001130968,0.0011093651,0.0024944867,0.0031260995,0.0031345733,0.0036470057,0.00077771867],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010900054,0.00006482615,0.0002805455,0.000084391024,0.00004644152,0.00004872984,0.000047266174,0.9459049,0.00016194695,0.0390145,0.0008459356,0.013391489],"study_design_scores_gemma":[0.000017987784,0.000011423496,0.00002561497,0.00000379062,0.000004134504,0.0000034601653,0.0000022137033,0.9873132,0.000029390936,0.0125119,0.00007311369,0.0000036879806],"about_ca_topic_score_codex":0.0097804675,"about_ca_topic_score_gemma":0.008133859,"teacher_disagreement_score":0.0097804675,"about_ca_system_score_codex":0.002711039,"about_ca_system_score_gemma":0.0028326227,"threshold_uncertainty_score":0.03527105},"labels":[],"label_agreement":null},{"id":"W1982435807","doi":"10.1142/s0218488511007416","title":"MULTIAGENT EXPEDITION WITH GRAPHICAL MODELS","year":2011,"lang":"en","type":"article","venue":"International Journal of Uncertainty Fuzziness and Knowledge-Based Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Scalability; Computer science; Graphical model; Set (abstract data type); Class (philosophy); Observable; Markov decision process; Multi-agent system; Artificial intelligence; Partially observable Markov decision process; Markov chain; Mathematical optimization; Markov process; Machine learning; Markov model; Mathematics","score_opus":0.04116553772219097,"score_gpt":0.25485211612003783,"score_spread":0.21368657839784685,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1982435807","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020101763,0.00020563128,0.97413427,0.00044843124,0.00002333569,0.00005303877,0.00011771509,0.00029202498,0.004623762],"genre_scores_gemma":[0.8238895,0.0004170669,0.17073464,0.00012858426,0.000031143813,0.00021906986,0.000239705,0.00007201307,0.0042683342],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99916625,0.0003936049,0.000031975494,0.00015796452,0.00015502138,0.000095344905],"domain_scores_gemma":[0.9978503,0.0014441661,0.00027559014,0.00017966493,0.000120072604,0.00013029942],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008398872,0.00088650524,0.0009231054,0.00060869986,0.000513799,0.0012615727,0.0014116727,0.0013094441,0.002671626],"category_scores_gemma":[0.0035867877,0.000548793,0.0011066019,0.000779635,0.0015879605,0.001892586,0.0016407657,0.0016158342,0.00032168883],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000025464466,0.000020969823,0.0002475688,0.000029539071,0.000016763384,0.00006491017,0.000042741773,0.9455101,0.00036688172,0.049040016,0.00028394812,0.004351221],"study_design_scores_gemma":[0.000010464946,0.000012962301,0.000046748537,0.0000041842122,0.000004643858,0.000011822441,0.000008219775,0.9633332,0.000142251,0.035896834,0.00052446284,0.000004202825],"about_ca_topic_score_codex":0.006228363,"about_ca_topic_score_gemma":0.0063795885,"teacher_disagreement_score":0.006228363,"about_ca_system_score_codex":0.0014007017,"about_ca_system_score_gemma":0.0010611033,"threshold_uncertainty_score":0.012384236},"labels":[],"label_agreement":null},{"id":"W1985646234","doi":"10.1115/imece2007-41643","title":"A Modified Q-Learning Algorithm for Multi-Robot Decision Making","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Robot; Computer science; Q-learning; Markov decision process; Artificial intelligence; Algorithm; Kalman filter; Robot learning; Domain (mathematical analysis); Markov chain; Reinforcement learning; Markov process; Machine learning; Mobile robot; Mathematics","score_opus":0.05076914814580656,"score_gpt":0.3375238673067552,"score_spread":0.28675471916094863,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1985646234","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0012553048,0.00011087835,0.9977012,0.000085621745,0.00004459646,0.00004227184,0.000009036659,0.00010046769,0.0006506765],"genre_scores_gemma":[0.26824442,0.00034158843,0.7261795,0.00034331557,0.00015783143,0.0005284455,0.00011453838,0.00008593832,0.0040044785],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980508,0.00069946656,0.00012154507,0.0004431322,0.00053087086,0.0001541855],"domain_scores_gemma":[0.9963689,0.0022365113,0.00023211246,0.0002199167,0.00081781857,0.00012486089],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028440387,0.0010764207,0.0014057879,0.000669419,0.0007216532,0.001083434,0.002555532,0.0020085825,0.004937784],"category_scores_gemma":[0.007320062,0.00048162494,0.0008223919,0.0008512358,0.0012090275,0.0017384604,0.0014956681,0.0022736262,0.001035689],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012654731,0.00009882555,0.000867978,0.00015020845,0.00007531621,0.00010686427,0.00012652246,0.76112664,0.0020684523,0.031248407,0.0022350387,0.2017692],"study_design_scores_gemma":[0.000024508861,0.00004299085,0.00005963767,0.0000065446243,0.0000053809767,0.000022442002,0.0000048546317,0.9920082,0.00034837122,0.0063746786,0.0010955305,0.000006771681],"about_ca_topic_score_codex":0.004502969,"about_ca_topic_score_gemma":0.0026692257,"teacher_disagreement_score":0.004937784,"about_ca_system_score_codex":0.001078435,"about_ca_system_score_gemma":0.0022174467,"threshold_uncertainty_score":0.016518593},"labels":[],"label_agreement":null},{"id":"W1986670254","doi":"10.1115/imece2007-41644","title":"Assess Team Q-Learning Algorithm in a Purely Cooperative Multi-Robot Task","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Robot; Computer science; Task (project management); Artificial intelligence; Convergence (economics); Q-learning; Robot learning; Object (grammar); Robotics; Markov decision process; Algorithm; Machine learning; Markov process; Reinforcement learning; Mobile robot; Engineering","score_opus":0.0296036776211122,"score_gpt":0.29674846523905857,"score_spread":0.26714478761794636,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1986670254","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022559164,0.00031174216,0.97413594,0.00027864604,0.000054937354,0.00010583526,0.000021778358,0.00020857925,0.002323302],"genre_scores_gemma":[0.8015011,0.00034382314,0.19328842,0.00032832433,0.00007856432,0.00043251592,0.00013851296,0.00006898374,0.003819769],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99868757,0.0004616087,0.000079276564,0.00031662092,0.00024817523,0.00020670106],"domain_scores_gemma":[0.99549013,0.0030240056,0.00031971757,0.0001293213,0.00078915094,0.00024767214],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00337509,0.00091663806,0.0015812931,0.0005677325,0.0008000378,0.001020147,0.0017403532,0.0016908658,0.0028164082],"category_scores_gemma":[0.006918939,0.00039204513,0.0005308312,0.00053176226,0.0009834054,0.0013812035,0.0015831573,0.001516459,0.00040127785],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014419209,0.0000807079,0.0011823677,0.000090882284,0.000043332813,0.00006695771,0.00009858125,0.933744,0.00059975847,0.008526169,0.0010449992,0.054378077],"study_design_scores_gemma":[0.000020683965,0.0000397952,0.00006372337,0.000003550762,0.0000050799786,0.000008142244,0.000008438542,0.9975266,0.000118334756,0.0020359661,0.0001664874,0.0000031900086],"about_ca_topic_score_codex":0.0068261954,"about_ca_topic_score_gemma":0.0025165118,"teacher_disagreement_score":0.0068261954,"about_ca_system_score_codex":0.0009859735,"about_ca_system_score_gemma":0.0026984056,"threshold_uncertainty_score":0.017849386},"labels":[],"label_agreement":null},{"id":"W1990320219","doi":"10.1145/2739480.2754798","title":"Knowledge Transfer from Keepaway Soccer to Half-field Offense through Program Symbiosis","year":2015,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Testbed; Computer science; Artificial intelligence; Task (project management); Modular design; Transfer of learning; Function (biology); Genetic programming; Machine learning; Engineering","score_opus":0.0649594141187844,"score_gpt":0.3187345708745497,"score_spread":0.2537751567557653,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1990320219","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.70272017,0.00016593368,0.28497964,0.0006830673,0.00004265099,0.000099170684,0.00002572173,0.0007005851,0.010583081],"genre_scores_gemma":[0.98329115,0.000037738624,0.01554769,0.000060501778,0.000003677906,0.00003655556,0.000016424325,0.00003074666,0.00097557704],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99974495,0.00008572941,0.000009396691,0.000059318612,0.00004869041,0.000051872354],"domain_scores_gemma":[0.9991584,0.00042226183,0.00008697758,0.00013159977,0.000083898136,0.0001169618],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00078723225,0.00054161606,0.0003776975,0.00018573792,0.00030848832,0.00060656,0.0007074111,0.0007134206,0.0016823362],"category_scores_gemma":[0.004116881,0.00018428714,0.00024017277,0.00012719916,0.0010572623,0.00081441045,0.0021363443,0.0008271206,0.00017014636],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019353782,0.00028392707,0.002792533,0.000084359206,0.000036121313,0.00022539811,0.0005299544,0.88203067,0.010533895,0.017989008,0.00076164387,0.084538914],"study_design_scores_gemma":[0.000036185298,0.0002320496,0.0005390527,0.000014678537,0.000014022529,0.000046472633,0.00014339747,0.9773303,0.0037655006,0.016694566,0.0011736052,0.000010113196],"about_ca_topic_score_codex":0.0015300994,"about_ca_topic_score_gemma":0.00081738085,"teacher_disagreement_score":0.0016823362,"about_ca_system_score_codex":0.0004926517,"about_ca_system_score_gemma":0.0007033036,"threshold_uncertainty_score":0.00562799},"labels":[],"label_agreement":null},{"id":"W1998172110","doi":"10.1007/s10479-005-5732-z","title":"Basis Function Adaptation in Temporal Difference Reinforcement Learning","year":2005,"lang":"en","type":"article","venue":"Annals of Operations Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":190,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Markov decision process; Bellman equation; Temporal difference learning; Basis function; Function approximation; Mathematical optimization; Basis (linear algebra); Computer science; Theory of computation; Q-learning; Convergence (economics); Markov process; Mathematics; Artificial intelligence; Algorithm; Artificial neural network","score_opus":0.22553966350065155,"score_gpt":0.41189285311802815,"score_spread":0.1863531896173766,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1998172110","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029049708,0.00060603366,0.9670018,0.00036145447,0.00012617817,0.000021962138,0.000021535607,0.00009687037,0.002714414],"genre_scores_gemma":[0.88729393,0.00065115234,0.1028082,0.00019248761,0.00011745542,0.00017652474,0.00006666827,0.00008739467,0.008606263],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99950457,0.00024576212,0.000024539992,0.00007739335,0.00009781486,0.000049900995],"domain_scores_gemma":[0.9972149,0.0021732536,0.000103482904,0.000116297924,0.00030950175,0.00008235032],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021569652,0.00046280693,0.0010366007,0.00029961832,0.00031064515,0.00080354477,0.0011975305,0.0012773193,0.0024812284],"category_scores_gemma":[0.007963539,0.0004517107,0.0003984519,0.00048199922,0.0011567572,0.0014104607,0.0010041799,0.0017932414,0.00024479488],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014934162,0.00009672557,0.00050531497,0.00007805336,0.00005136018,0.00004578877,0.0000649386,0.8295184,0.0015981379,0.103106715,0.0012621916,0.063523024],"study_design_scores_gemma":[0.000007551465,0.000012140875,0.00003912542,0.0000023088326,0.0000031434936,0.000004201376,0.0000019761324,0.98559546,0.00010637573,0.014095197,0.0001297879,0.0000027532728],"about_ca_topic_score_codex":0.0037365952,"about_ca_topic_score_gemma":0.0020765471,"teacher_disagreement_score":0.0037365952,"about_ca_system_score_codex":0.0007637889,"about_ca_system_score_gemma":0.00084502716,"threshold_uncertainty_score":0.0114071965},"labels":[],"label_agreement":null},{"id":"W1998314229","doi":"10.1145/1089827.1089829","title":"Reinforcement learning for active model selection","year":2005,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Markov decision process; Artificial intelligence; Machine learning; Learning classifier system; Classifier (UML); Feature selection; Selection (genetic algorithm); Training set; Feature (linguistics); Active learning (machine learning); Markov process; Mathematics","score_opus":0.02332588564199617,"score_gpt":0.27049509625491275,"score_spread":0.24716921061291658,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1998314229","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007511758,0.00031796837,0.9893056,0.0003590946,0.000051249946,0.0000642865,0.000026181853,0.00031628952,0.0020476037],"genre_scores_gemma":[0.81508523,0.00037972617,0.1783703,0.00047352063,0.00015997676,0.0006348022,0.00015257129,0.0001222552,0.004621635],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984836,0.0008149971,0.0000659093,0.00021943069,0.00028650725,0.00012957442],"domain_scores_gemma":[0.9908936,0.007339269,0.00045097645,0.0004258706,0.0006340723,0.0002561991],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004163533,0.0013533088,0.0019444495,0.0006870073,0.0005828917,0.0011346228,0.0023737405,0.001838744,0.004191409],"category_scores_gemma":[0.014483242,0.0006964311,0.0006027406,0.00058686634,0.002009049,0.001759668,0.0015872802,0.00282885,0.0007385939],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013823133,0.0001357906,0.000739051,0.00010456392,0.000060948045,0.00009188858,0.0000812524,0.8906666,0.00076975895,0.05137312,0.0019469488,0.053891893],"study_design_scores_gemma":[0.000019547324,0.000020981619,0.000025278296,0.000005879429,0.000003964396,0.000007920907,0.000003140684,0.9877176,0.0001666701,0.011743387,0.00028213792,0.000003392267],"about_ca_topic_score_codex":0.0025761703,"about_ca_topic_score_gemma":0.002582131,"teacher_disagreement_score":0.004191409,"about_ca_system_score_codex":0.0014077646,"about_ca_system_score_gemma":0.001315198,"threshold_uncertainty_score":0.022019148},"labels":[],"label_agreement":null},{"id":"W1999095902","doi":"10.1109/cig.2013.6633642","title":"Stacked calibration of off-policy policy evaluation for video game matchmaking","year":2013,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ubisoft (Canada); Université de Montréal","funders":"","keywords":"Computer science; Function (biology); Calibration; Video game; Simple (philosophy); Statistics; Multimedia; Mathematics","score_opus":0.03258298194827107,"score_gpt":0.32181601398908705,"score_spread":0.28923303204081596,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1999095902","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11397191,0.00063710386,0.87774193,0.00046942584,0.00014399088,0.00021111868,0.00016988146,0.003316715,0.003337802],"genre_scores_gemma":[0.9028541,0.00013174486,0.09472997,0.00024136566,0.000045607507,0.00017188097,0.00029511855,0.0001910048,0.0013391753],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99704945,0.0011942732,0.00016056508,0.0007483484,0.0005046161,0.00034265712],"domain_scores_gemma":[0.98707706,0.007958332,0.0012236786,0.0017259137,0.0015113428,0.0005037163],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0078012706,0.0019737564,0.0018805463,0.0010862827,0.0006688668,0.0016327718,0.0023980741,0.0028200813,0.0027352674],"category_scores_gemma":[0.044058535,0.0012110105,0.0006730965,0.00075910863,0.0014056092,0.0027047817,0.002039242,0.00399794,0.0009577582],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002899613,0.00029311862,0.0033801305,0.000080907674,0.00010530766,0.00005084766,0.00009472878,0.9086015,0.002928053,0.0027806077,0.0010027878,0.080392085],"study_design_scores_gemma":[0.000011840724,0.00006899599,0.00059705187,0.000012472471,0.000008279768,0.000013902612,0.000010919938,0.9954906,0.0015426629,0.0020243875,0.000204328,0.000014632572],"about_ca_topic_score_codex":0.008292704,"about_ca_topic_score_gemma":0.0066478956,"teacher_disagreement_score":0.008292704,"about_ca_system_score_codex":0.0021117297,"about_ca_system_score_gemma":0.0024637042,"threshold_uncertainty_score":0.0412575},"labels":[],"label_agreement":null},{"id":"W2001487393","doi":"10.1155/2009/984752","title":"Vector Field Driven Design for Lightweight Signal Processing and Control Schemes for Autonomous Robotic Navigation","year":2009,"lang":"en","type":"article","venue":"EURASIP Journal on Advances in Signal Processing","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Neuromorphic engineering; Robotics; Artificial intelligence; Field (mathematics); Exploit; Signal processing; SIGNAL (programming language); Computer architecture; Control engineering; Robot; Human–computer interaction; Digital signal processing; Computer hardware; Artificial neural network","score_opus":0.019334171976119554,"score_gpt":0.29524634823391543,"score_spread":0.2759121762577959,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2001487393","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0067992928,0.000083019484,0.99045724,0.00011525235,0.000028011475,0.0000340221,0.000007990255,0.00008497916,0.0023902226],"genre_scores_gemma":[0.5335156,0.00030921694,0.45969483,0.00014966636,0.00005279872,0.00030097863,0.000040089577,0.000071426235,0.0058653047],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99982005,0.000048462796,0.000011633889,0.00002187917,0.00008336267,0.0000145925],"domain_scores_gemma":[0.9997768,0.000093708455,0.000038475886,0.000029932946,0.00004614848,0.00001499677],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006003787,0.000327903,0.000239574,0.00027064708,0.000264055,0.0005633485,0.0005384283,0.00046618615,0.0025779475],"category_scores_gemma":[0.000920514,0.0001833124,0.00029183127,0.00020655595,0.0006225312,0.0007229401,0.00053181144,0.0005726148,0.0003504772],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000063783285,0.00006180745,0.00023785449,0.00019100338,0.000019543473,0.00008913166,0.00017908642,0.31914726,0.03844903,0.5581592,0.0012499309,0.0821523],"study_design_scores_gemma":[0.000033736935,0.00011297876,0.000057713376,0.00001964052,0.000006635687,0.00004055442,0.000015448017,0.9086676,0.004967377,0.079166874,0.0068983254,0.000013097138],"about_ca_topic_score_codex":0.0003092731,"about_ca_topic_score_gemma":0.00043446833,"teacher_disagreement_score":0.0025779475,"about_ca_system_score_codex":0.00051286956,"about_ca_system_score_gemma":0.00042890946,"threshold_uncertainty_score":0.008624136},"labels":[],"label_agreement":null},{"id":"W2002256627","doi":"10.1109/adprl.2013.6614986","title":"Exponential moving average Q-learning algorithm","year":2013,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Reinforcement learning; Variety (cybernetics); Computer science; Q-learning; Exponential function; Algorithm; Policy learning; Markov decision process; State (computer science); Mathematical optimization; Matrix (chemical analysis); Artificial intelligence; Machine learning; Mathematics; Markov process","score_opus":0.007682844823853384,"score_gpt":0.2071727335532115,"score_spread":0.1994898887293581,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2002256627","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0028522536,0.00019686866,0.9948639,0.00013844387,0.000041548057,0.000047482597,0.000014739796,0.00019751594,0.001647235],"genre_scores_gemma":[0.47906685,0.0004746602,0.5099707,0.00045602914,0.00013186295,0.0004896005,0.00018779021,0.00011109968,0.009111368],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99858403,0.00054697925,0.00007592537,0.00029111333,0.00035297012,0.00014898847],"domain_scores_gemma":[0.99751365,0.0013538247,0.00018299234,0.00016255108,0.0006764848,0.00011052592],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025371776,0.0008191015,0.0017671442,0.0006691878,0.0006814266,0.0010996258,0.002371465,0.0015665494,0.004654617],"category_scores_gemma":[0.0065091583,0.00044904783,0.0005355998,0.0007714968,0.001043634,0.0012585046,0.0015385052,0.0016604697,0.0011212901],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010482534,0.000116746334,0.00079205324,0.000104514,0.00006750815,0.000076681696,0.00007423206,0.7893979,0.0010689105,0.04300994,0.0028360272,0.16235067],"study_design_scores_gemma":[0.000017769971,0.000023249399,0.00004223277,0.0000045650513,0.0000049766086,0.000016196926,0.0000033703168,0.9935324,0.00021838593,0.0054308535,0.00070186116,0.000004137365],"about_ca_topic_score_codex":0.0038393438,"about_ca_topic_score_gemma":0.0020069736,"teacher_disagreement_score":0.004654617,"about_ca_system_score_codex":0.0008904142,"about_ca_system_score_gemma":0.0022991628,"threshold_uncertainty_score":0.015571237},"labels":[],"label_agreement":null},{"id":"W200360931","doi":"","title":"Turning lights out with DQ-learning","year":2006,"lang":"en","type":"article","venue":"International conference on Artificial intelligence and applications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Discretization; Premise; Arity; Overhead (engineering); Byte; Table (database); Representation (politics); Reinforcement learning; Artificial intelligence; State (computer science); Algorithm; Theoretical computer science; Mathematics; Discrete mathematics; Programming language; Database","score_opus":0.06063246829204915,"score_gpt":0.308802611147818,"score_spread":0.24817014285576883,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W200360931","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0075676707,0.00027283537,0.98440003,0.00150111,0.00017469193,0.00003034738,0.000026583066,0.00046301045,0.0055637006],"genre_scores_gemma":[0.50517243,0.0005541239,0.48170352,0.002075983,0.00029515108,0.00021759694,0.00012413847,0.00029434747,0.009562753],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99864787,0.0005859792,0.000065331595,0.00027389408,0.00032040183,0.00010652084],"domain_scores_gemma":[0.9961099,0.0022431994,0.00018869771,0.00077770476,0.00047693218,0.00020364171],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023155988,0.0005457282,0.0009947645,0.000279933,0.000564226,0.0013497646,0.0014181357,0.0013086344,0.0058294698],"category_scores_gemma":[0.013224048,0.00038001584,0.0005167843,0.00040341675,0.0024958302,0.0033277597,0.0022567003,0.0032182902,0.0010821503],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027596383,0.00014353698,0.0014445893,0.0002171667,0.0000935095,0.00012138673,0.00026709682,0.29660314,0.001457285,0.5000557,0.011452072,0.18786862],"study_design_scores_gemma":[0.000066839544,0.00007159984,0.00010243818,0.000025535262,0.0000107276155,0.000029920498,0.000041560867,0.65079606,0.00077580486,0.34116268,0.006901978,0.000014912836],"about_ca_topic_score_codex":0.00228812,"about_ca_topic_score_gemma":0.0015211398,"teacher_disagreement_score":0.0058294698,"about_ca_system_score_codex":0.00093125744,"about_ca_system_score_gemma":0.0013403657,"threshold_uncertainty_score":0.019501567},"labels":[],"label_agreement":null},{"id":"W2006330826","doi":"10.1007/s10994-011-5254-7","title":"Model selection in reinforcement learning","year":2011,"lang":"en","type":"article","venue":"Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Mathematics; Estimator; Oracle; Bellman equation; Function (biology); Regularization (linguistics); Countable set; Sequence (biology); Combinatorics; Rate of convergence; Discrete mathematics; Mathematical optimization; Computer science; Artificial intelligence; Statistics","score_opus":0.03659510899503107,"score_gpt":0.24580997705280483,"score_spread":0.20921486805777376,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2006330826","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012220145,0.0020213109,0.9801543,0.0010141563,0.00014141166,0.000038213548,0.000033490716,0.00020341306,0.0041735917],"genre_scores_gemma":[0.849001,0.0014213797,0.13946238,0.00045874843,0.0003252094,0.0003296143,0.0001346998,0.0001556336,0.0087113585],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980781,0.0013385181,0.00005620794,0.00019279431,0.00024473827,0.00008965623],"domain_scores_gemma":[0.9912907,0.0077144084,0.00024140437,0.00025223292,0.00036835176,0.00013287444],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030007774,0.00081852224,0.0019548861,0.0006059599,0.0004927586,0.0013216038,0.0015054166,0.0018147297,0.0036415898],"category_scores_gemma":[0.013706853,0.00077420194,0.00061952154,0.0007234466,0.0019986625,0.0020152463,0.0012693251,0.0023956492,0.00044267968],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009259565,0.0000693143,0.0006326678,0.00013829934,0.000082804865,0.00007131924,0.00006421772,0.8254107,0.00031442178,0.12656851,0.0024449916,0.044110075],"study_design_scores_gemma":[0.000018164917,0.000014721206,0.000045204397,0.000007917492,0.0000074739883,0.00000790178,0.0000033212275,0.94519067,0.00009519334,0.054198172,0.00040718156,0.000004069874],"about_ca_topic_score_codex":0.0032973285,"about_ca_topic_score_gemma":0.0022385558,"teacher_disagreement_score":0.0036415898,"about_ca_system_score_codex":0.0013177665,"about_ca_system_score_gemma":0.0008361903,"threshold_uncertainty_score":0.015869856},"labels":[],"label_agreement":null},{"id":"W2008768326","doi":"10.1023/b:jotp.0000011995.28536.ef","title":"Continuity of the Value of Competitive Markov Decision Processes","year":2003,"lang":"en","type":"article","venue":"Journal of Theoretical Probability","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kellogg's (Canada)","funders":"","keywords":"Mathematics; Markov decision process; Markov chain; Upper and lower bounds; Markov process; Markov kernel; Discounting; Value (mathematics); Mathematical optimization; Partially observable Markov decision process; Space (punctuation); Bellman equation; Function (biology); Markov model; Mathematical economics; Statistics; Variable-order Markov model; Computer science; Mathematical analysis; Economics","score_opus":0.008738568404383095,"score_gpt":0.247020988068592,"score_spread":0.2382824196642089,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2008768326","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.30597517,0.0021507065,0.63357365,0.0052428152,0.00017554442,0.00010216415,0.00039873197,0.00023540898,0.052145883],"genre_scores_gemma":[0.9722325,0.0007461302,0.020497002,0.00017046812,0.00011498975,0.000078708246,0.000100347504,0.00004614944,0.006013715],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9977436,0.0010493307,0.00010023977,0.0003166715,0.00044313943,0.0003469444],"domain_scores_gemma":[0.966453,0.028294016,0.0013525151,0.00091918895,0.0013867306,0.0015944716],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004969254,0.00064091344,0.0014975751,0.0012787569,0.0009859354,0.003921583,0.002027475,0.002188517,0.006535301],"category_scores_gemma":[0.035966553,0.00085973705,0.00092390785,0.0010060669,0.0037267597,0.0062438827,0.001972456,0.0033872584,0.00039654152],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000102922175,0.00003500745,0.0004514025,0.00005557702,0.000022926943,0.00006705512,0.00012609902,0.02619076,0.00038186635,0.9668756,0.0004995339,0.005191265],"study_design_scores_gemma":[0.00003117146,0.000034506556,0.00024076033,0.000023083023,0.000011086049,0.000038082337,0.000033869743,0.15124112,0.00019291507,0.8474592,0.0006793367,0.000014825977],"about_ca_topic_score_codex":0.0020950064,"about_ca_topic_score_gemma":0.0010981284,"teacher_disagreement_score":0.006535301,"about_ca_system_score_codex":0.0027265225,"about_ca_system_score_gemma":0.0016832751,"threshold_uncertainty_score":0.026280224},"labels":[],"label_agreement":null},{"id":"W2010124866","doi":"10.2316/journal.206.2006.2.206-2795","title":"FUZZY REINFORCEMENT LEARNING FOR EMBEDDED SOCCER AGENTS IN A MULTI-AGENT CONTEXT","year":2006,"lang":"en","type":"article","venue":"International Journal of Robotics and Automation","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Fuzzy logic; Context (archaeology); Artificial intelligence; Reinforcement; Machine learning; Engineering","score_opus":0.03483910685198837,"score_gpt":0.30849806606312313,"score_spread":0.27365895921113476,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2010124866","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09526001,0.00026610913,0.89996153,0.00035425057,0.00005384472,0.000047105317,0.000018130331,0.00019631821,0.0038427473],"genre_scores_gemma":[0.9685787,0.0000748027,0.029645415,0.000031536558,0.000013821058,0.000042642194,0.000011046095,0.000009976366,0.0015920583],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99965036,0.0001391224,0.000017196757,0.00005990078,0.00008403989,0.00004945664],"domain_scores_gemma":[0.99916327,0.00043681767,0.000128987,0.000044532353,0.00014359968,0.00008270802],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008587005,0.00049815874,0.00062701653,0.00024716524,0.00049744104,0.000837882,0.000851336,0.00082278496,0.0013884632],"category_scores_gemma":[0.002327996,0.00022696069,0.0003243805,0.00015839997,0.0010134373,0.0007813843,0.0008278645,0.00078196905,0.00013635386],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006561549,0.000034169178,0.00030303426,0.000031216034,0.000023240746,0.000101780504,0.000082337785,0.97363925,0.0016634966,0.014198597,0.00014563459,0.009711547],"study_design_scores_gemma":[0.00000818043,0.000018051478,0.000038531165,0.000002190415,0.0000024908225,0.000006007908,0.0000068464083,0.99601054,0.0001937211,0.0035822089,0.00012890757,0.0000022867298],"about_ca_topic_score_codex":0.0063441014,"about_ca_topic_score_gemma":0.0039697047,"teacher_disagreement_score":0.0063441014,"about_ca_system_score_codex":0.00094035774,"about_ca_system_score_gemma":0.00079161715,"threshold_uncertainty_score":0.012614369},"labels":[],"label_agreement":null},{"id":"W2011033111","doi":"10.1109/adprl.2007.368201","title":"Opposition-Based Q(&amp;#x003BB;) with Non-Markovian Update","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Opposition (politics); Computer science; Markov process; Algorithm; Theoretical computer science; Statistical physics; Mathematics; Physics; Law; Statistics","score_opus":0.009002528772758794,"score_gpt":0.23877252817285816,"score_spread":0.22976999940009937,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2011033111","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008023033,0.00006622076,0.9893561,0.00008116649,0.00004794302,0.00008512123,0.000018642353,0.0004815573,0.0018403176],"genre_scores_gemma":[0.4063289,0.00012675511,0.5864031,0.00032703608,0.00006107147,0.0003357237,0.00011273618,0.00019752672,0.0061071157],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99893767,0.00032032616,0.000065602944,0.00015983173,0.00041797376,0.00009863339],"domain_scores_gemma":[0.9967803,0.0020777616,0.00021707056,0.00029275377,0.0004895858,0.00014247106],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015995678,0.0005190344,0.00071123004,0.0006648856,0.00044057937,0.00070679415,0.0021462126,0.00094447593,0.004873447],"category_scores_gemma":[0.00673325,0.0003051132,0.00050146074,0.00067667005,0.0008555293,0.0013322841,0.0015124126,0.001223426,0.0007321712],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045490492,0.00041291935,0.0015497815,0.00024289837,0.0000765867,0.00015120528,0.00022880512,0.1845423,0.008106159,0.059480604,0.0045330836,0.74022067],"study_design_scores_gemma":[0.00007534815,0.00018334546,0.00030213947,0.000012172503,0.000016807013,0.00009428253,0.00002039094,0.9761562,0.0033573834,0.01665371,0.0031115846,0.000016559576],"about_ca_topic_score_codex":0.0042313566,"about_ca_topic_score_gemma":0.004601266,"teacher_disagreement_score":0.004873447,"about_ca_system_score_codex":0.000636614,"about_ca_system_score_gemma":0.0017872807,"threshold_uncertainty_score":0.0163033},"labels":[],"label_agreement":null},{"id":"W2011231614","doi":"10.1145/1329125.1329178","title":"Reducing the complexity of multiagent reinforcement learning","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Reinforcement learning; Initialization; Computer science; Function (biology); Artificial intelligence; Representation (politics); Class (philosophy); Multi-agent system; Function approximation; Mathematical optimization; Sample complexity; Computational complexity theory; Action (physics); Mathematics; Algorithm; Artificial neural network","score_opus":0.060374917235600525,"score_gpt":0.28898364382667147,"score_spread":0.22860872659107095,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2011231614","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04247697,0.000230166,0.95204633,0.0006722909,0.000041388277,0.000079729085,0.000032979555,0.00051083264,0.003909223],"genre_scores_gemma":[0.856557,0.00017651936,0.13999026,0.00016453695,0.000050889474,0.00026571951,0.00007937842,0.000101309706,0.0026143808],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99840385,0.0007108953,0.00007226803,0.00020143743,0.00045211866,0.0001593948],"domain_scores_gemma":[0.99325925,0.0049706856,0.00041197997,0.00070361554,0.00046632387,0.00018814873],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023993968,0.0010601651,0.0014426206,0.0004067844,0.0005664998,0.0010405311,0.0014197634,0.0012023866,0.0020563214],"category_scores_gemma":[0.01038532,0.00069374626,0.000715076,0.0002826195,0.00121772,0.0020371093,0.0022879122,0.0028984244,0.00033518774],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011001679,0.00005878848,0.00042232205,0.00005862388,0.00003120786,0.000057368132,0.00006818516,0.95862234,0.0010812139,0.02368331,0.00050984137,0.015296718],"study_design_scores_gemma":[0.000010552193,0.00001133042,0.000042777367,0.0000025533873,0.0000026913494,0.0000046385153,0.000003928825,0.99033886,0.00020943068,0.009225902,0.00014502331,0.0000023253203],"about_ca_topic_score_codex":0.004878637,"about_ca_topic_score_gemma":0.0035394633,"teacher_disagreement_score":0.004878637,"about_ca_system_score_codex":0.0018774585,"about_ca_system_score_gemma":0.0016320095,"threshold_uncertainty_score":0.013621986},"labels":[],"label_agreement":null},{"id":"W2011233848","doi":"10.1145/1143844.1143901","title":"Automatic basis function construction for approximate dynamic programming and reinforcement learning","year":2006,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":162,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Bellman equation; Markov decision process; Reinforcement learning; Computer science; Basis (linear algebra); Function approximation; Dynamic programming; Curse of dimensionality; Basis function; Mathematical optimization; State space; Temporal difference learning; Markov process; Dimensionality reduction; Q-learning; Function (biology); Space (punctuation); Artificial intelligence; Algorithm; Mathematics; Artificial neural network","score_opus":0.007772714293839559,"score_gpt":0.22287687197202508,"score_spread":0.21510415767818553,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2011233848","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002335824,0.000044678618,0.99711037,0.0000330998,0.000010986024,0.000011409612,0.000007413537,0.00013837458,0.00030793046],"genre_scores_gemma":[0.20302154,0.00018587275,0.7945531,0.000070547925,0.000042672807,0.00033713476,0.00012291498,0.00022943533,0.0014367251],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998953,0.00045541904,0.000047151556,0.00013926103,0.00032539418,0.000079839585],"domain_scores_gemma":[0.9979037,0.0011761906,0.00014478675,0.0003154153,0.000389514,0.00007047344],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020725816,0.0007558373,0.0015000327,0.00087959616,0.0005583897,0.0009049407,0.0014626733,0.0013872075,0.0021915515],"category_scores_gemma":[0.0074001555,0.0006413195,0.00083072315,0.0007980954,0.0009992112,0.0016074902,0.0015566347,0.0022370962,0.00084150495],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006899658,0.00011502628,0.0005144865,0.000117084855,0.000045358673,0.00006251086,0.0001120284,0.69763786,0.004512556,0.103428915,0.0014019512,0.19198333],"study_design_scores_gemma":[0.0000036402416,0.000009327377,0.000013840439,0.000003508206,0.0000013719111,0.0000051168863,0.0000020332575,0.9867373,0.00048220405,0.01246651,0.0002718017,0.0000033100807],"about_ca_topic_score_codex":0.0021438776,"about_ca_topic_score_gemma":0.0019102569,"teacher_disagreement_score":0.0021915515,"about_ca_system_score_codex":0.00076947367,"about_ca_system_score_gemma":0.0010323889,"threshold_uncertainty_score":0.010960996},"labels":[],"label_agreement":null},{"id":"W2012045703","doi":"10.1287/moor.27.3.545.316","title":"Achieving Target State-Action Frequencies in Multichain Average-Reward Markov Decision Processes","year":2002,"lang":"en","type":"article","venue":"Mathematics of Operations Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Markov decision process; Stochastic game; Construct (python library); Mathematics; Action (physics); Mathematical optimization; Markov process; State space; Markov chain; State (computer science); Space (punctuation); Decision problem; Mathematical economics; Computer science; Algorithm; Statistics","score_opus":0.12878878589770923,"score_gpt":0.37332200477531274,"score_spread":0.24453321887760351,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2012045703","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2087307,0.00017489302,0.7884183,0.000301561,0.000014141667,0.00006841552,0.000072181225,0.00018416395,0.0020357107],"genre_scores_gemma":[0.89820534,0.0001504529,0.10024232,0.00006252663,0.000013875549,0.00013725519,0.000090234345,0.00003925787,0.0010586558],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9985216,0.00058386417,0.000078744946,0.00037739964,0.0002367002,0.00020154881],"domain_scores_gemma":[0.99592304,0.0028954926,0.00060643844,0.00015940717,0.00018643642,0.00022916833],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030989072,0.0007964407,0.001164758,0.0006363628,0.0007850661,0.0010541255,0.0010127053,0.0012232076,0.0013417241],"category_scores_gemma":[0.007434989,0.0005727917,0.00055462815,0.0007543579,0.0019887555,0.0023764838,0.0014877078,0.0012782217,0.00019101796],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001010164,0.0000684611,0.00069272757,0.000046748253,0.000024719322,0.00007264789,0.00011412214,0.91979825,0.00087418867,0.06785767,0.00013140889,0.010218093],"study_design_scores_gemma":[0.000026203872,0.000046859353,0.00015650681,0.000008673303,0.000006543357,0.000012655363,0.00001698696,0.9389081,0.00054753793,0.060093634,0.00016734634,0.000008917876],"about_ca_topic_score_codex":0.004738723,"about_ca_topic_score_gemma":0.004171516,"teacher_disagreement_score":0.004738723,"about_ca_system_score_codex":0.0017105337,"about_ca_system_score_gemma":0.0016304293,"threshold_uncertainty_score":0.016388774},"labels":[],"label_agreement":null},{"id":"W2014744886","doi":"10.1109/cig.2006.311676","title":"Grid-Robot Drivers: an Evolutionary Multi-agent Virtual Robotics Task","year":2006,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"University of Guelph","keywords":"Computer science; Artificial intelligence; Robot; Grid; Context (archaeology); Prisoner's dilemma; Task (project management); Reinforcement learning; Dilemma; Simulation; Engineering","score_opus":0.019255348371642514,"score_gpt":0.24418496478241183,"score_spread":0.2249296164107693,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2014744886","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8072928,0.000072008086,0.18726371,0.00032921822,0.000025868407,0.00013991894,0.000047504745,0.00027371972,0.004555131],"genre_scores_gemma":[0.92750823,0.000040498424,0.07031377,0.000037878213,0.0000030339625,0.00008459792,0.00004775486,0.000027096607,0.0019372896],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99969697,0.00017533799,0.000009347976,0.000042301413,0.000039427487,0.000036733254],"domain_scores_gemma":[0.9992544,0.00042577795,0.000058303383,0.0000997722,0.00004452911,0.00011724454],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00089340995,0.00044563643,0.00050816935,0.00018997537,0.00067222083,0.00066336564,0.0009000821,0.00084826146,0.001449328],"category_scores_gemma":[0.0021540083,0.000272821,0.0003732065,0.00013505972,0.0008318162,0.00090072636,0.0012013803,0.00054563535,0.00014515345],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042282342,0.0002689714,0.0044351383,0.00009553418,0.00008104278,0.0004781928,0.000741408,0.91751146,0.010370322,0.014740401,0.0008039535,0.05005075],"study_design_scores_gemma":[0.00008877568,0.00025039364,0.0006377145,0.0000065975782,0.000016232852,0.00010415358,0.00024294038,0.9844349,0.003584858,0.008632421,0.0019835678,0.000017450504],"about_ca_topic_score_codex":0.0024675815,"about_ca_topic_score_gemma":0.002192752,"teacher_disagreement_score":0.0024675815,"about_ca_system_score_codex":0.00036164987,"about_ca_system_score_gemma":0.00062093197,"threshold_uncertainty_score":0.004906416},"labels":[],"label_agreement":null},{"id":"W2023718460","doi":"10.3166/ria.20.203-234","title":"Prise de décision en temps-réel pour des POMDP de grande taille","year":2006,"lang":"fr","type":"article","venue":"Revue d intelligence artificielle","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.042396513919707404,"score_gpt":0.28318727660601534,"score_spread":0.24079076268630795,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2023718460","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012278183,0.00022878913,0.98534346,0.00022215431,0.00003059469,0.000049531856,0.00007341427,0.00024488044,0.0015290702],"genre_scores_gemma":[0.52083534,0.00053664955,0.47236642,0.00014508267,0.00006985949,0.0003141554,0.00022186429,0.00013017969,0.0053804037],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99886346,0.0003658575,0.00005039018,0.0002594432,0.00031765067,0.00014324712],"domain_scores_gemma":[0.997855,0.0015198597,0.00020349346,0.000103485945,0.00016550197,0.00015264098],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020565572,0.0011656753,0.0013290003,0.00053104706,0.00064288796,0.0016380443,0.0011080798,0.0012214554,0.0025220455],"category_scores_gemma":[0.0045822957,0.00061357475,0.0009770185,0.000434904,0.0011075347,0.0015602383,0.0010168275,0.002772728,0.000418102],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027882901,0.000096698575,0.0009332261,0.0001272476,0.000052246218,0.000106159365,0.00011941341,0.91165113,0.0013949256,0.041747183,0.0009803498,0.042512506],"study_design_scores_gemma":[0.000023560384,0.00005173622,0.000101719284,0.000011303566,0.000008761115,0.00001912746,0.000014634362,0.99014974,0.0006141843,0.00819283,0.00080343435,0.000008932412],"about_ca_topic_score_codex":0.0067926017,"about_ca_topic_score_gemma":0.008967862,"teacher_disagreement_score":0.0067926017,"about_ca_system_score_codex":0.0020504703,"about_ca_system_score_gemma":0.002420656,"threshold_uncertainty_score":0.014877319},"labels":[],"label_agreement":null},{"id":"W2024074370","doi":"10.3166/ria.20.275-310","title":"Apprentissage par renforcement dans le cadre des processus décisionnels de Markov factorisés observables dans le désordre. Etude expérimentale du Q-Learning parallèle appliqué aux problèmes du labyrinthe et du New York Driving","year":2006,"lang":"fr","type":"article","venue":"Revue d intelligence artificielle","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Markov decision process; Observable; Architecture; Computer science; Markov chain; Partially observable Markov decision process; Humanities; Markov process; Artificial intelligence; Markov model; Mathematics; Physics; Machine learning; Philosophy; Art; Statistics","score_opus":0.03862456106017156,"score_gpt":0.2586800958815554,"score_spread":0.22005553482138382,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2024074370","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6734592,0.00014499112,0.32367975,0.00012918598,0.00003405589,0.00006832625,0.00004325926,0.0006587832,0.0017825295],"genre_scores_gemma":[0.9679354,0.000037185426,0.030898994,0.000011851717,0.0000036903382,0.000028004006,0.000024475825,0.000021269854,0.0010391839],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995883,0.00012207656,0.000016248241,0.00013397887,0.00007910175,0.00006026491],"domain_scores_gemma":[0.9986325,0.0007715085,0.00013336491,0.00015834243,0.00018674173,0.000117515796],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009750331,0.00054782064,0.00055379723,0.00023696016,0.00037934296,0.0007307394,0.0005464952,0.00072033465,0.0015535671],"category_scores_gemma":[0.0033381537,0.00033228227,0.0003479062,0.00022508325,0.0007997223,0.0010733006,0.00046299133,0.0008263253,0.00018157055],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051206397,0.00022277299,0.0022558486,0.000057928435,0.000042975666,0.00012039887,0.00017201957,0.91035044,0.027073115,0.006790704,0.00024773984,0.052153923],"study_design_scores_gemma":[0.000019316101,0.000079115845,0.00038276278,0.0000013562963,0.000004576444,0.000010896312,0.000011063768,0.99347526,0.004651247,0.0011765702,0.00018273221,0.000004997019],"about_ca_topic_score_codex":0.014734656,"about_ca_topic_score_gemma":0.008138643,"teacher_disagreement_score":0.014734656,"about_ca_system_score_codex":0.00096666906,"about_ca_system_score_gemma":0.00087305397,"threshold_uncertainty_score":0.02929777},"labels":[],"label_agreement":null},{"id":"W2027648864","doi":"10.2991/agi.2010.22","title":"GQ( ): A general gradient algorithm for temporal-difference prediction learning with eligibility traces","year":2010,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":109,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Algorithm; Temporal difference learning; Artificial intelligence; Reinforcement learning","score_opus":0.01308984256616065,"score_gpt":0.25791150519595524,"score_spread":0.2448216626297946,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2027648864","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0020110712,0.00007381991,0.9966607,0.00010701574,0.00003303821,0.00004822846,0.000026812757,0.00047898985,0.0005603284],"genre_scores_gemma":[0.15495016,0.00017804222,0.8397954,0.0003457774,0.000069554575,0.0004276026,0.00021117242,0.0003973737,0.0036249775],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99938524,0.00020949038,0.000044129156,0.0001325429,0.00015930305,0.00006938798],"domain_scores_gemma":[0.9988129,0.0006701143,0.00008467666,0.00011637481,0.00023419075,0.000081714374],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002528707,0.00090738194,0.0011139723,0.0007093922,0.00039737538,0.0010305081,0.0025184827,0.0018411322,0.005423053],"category_scores_gemma":[0.0069682617,0.0005336033,0.0006177147,0.00074731547,0.0010506,0.0020752857,0.002155966,0.0021133379,0.0010839687],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024906025,0.00014433305,0.0009947249,0.0001789367,0.00007416377,0.00009073278,0.00013417905,0.50948936,0.00242225,0.09726028,0.006282497,0.38267952],"study_design_scores_gemma":[0.000032172255,0.000030082534,0.000052389925,0.000008854149,0.0000054182938,0.000018076455,0.0000053136823,0.9791911,0.00043615303,0.01904161,0.0011724143,0.00000637876],"about_ca_topic_score_codex":0.006174657,"about_ca_topic_score_gemma":0.0046762973,"teacher_disagreement_score":0.006174657,"about_ca_system_score_codex":0.001133074,"about_ca_system_score_gemma":0.0021202515,"threshold_uncertainty_score":0.018141866},"labels":[],"label_agreement":null},{"id":"W2027757692","doi":"10.1080/10798587.2005.10642903","title":"Behavior Arbitration using a Fuzzy Reinforcement Learning Approach","year":2005,"lang":"en","type":"article","venue":"Intelligent Automation & Soft Computing","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Arbitration; Reinforcement learning; Fuzzy logic; Artificial intelligence; Law","score_opus":0.03712224588085534,"score_gpt":0.29007950211920536,"score_spread":0.25295725623835,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2027757692","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044179022,0.00011972712,0.95192987,0.00018169671,0.000031565123,0.00006911481,0.000010868407,0.00023112455,0.0032469297],"genre_scores_gemma":[0.9268141,0.00004230386,0.071624845,0.000058243633,0.000016743574,0.00007874369,0.000012692247,0.000013903673,0.0013383931],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99941087,0.00022127667,0.000033866596,0.00009653826,0.00016512071,0.00007239006],"domain_scores_gemma":[0.9986743,0.00069228176,0.00015098335,0.00008281984,0.00031326484,0.00008649911],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014929881,0.00046772626,0.00062756357,0.0004919382,0.00040329335,0.0005590663,0.0010161961,0.00077856216,0.0020079652],"category_scores_gemma":[0.002725413,0.0002073498,0.00040449132,0.00026596058,0.0008473247,0.00053884444,0.000615762,0.0008077472,0.00017182645],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010111338,0.00011508734,0.0008623307,0.00005164839,0.00004726627,0.00007811521,0.00008592379,0.9214429,0.0032367005,0.012526483,0.00042526377,0.06102719],"study_design_scores_gemma":[0.0000068482836,0.000018891524,0.00003779442,0.0000019763916,0.0000025304723,0.0000062912586,0.0000028753668,0.99829584,0.00025319596,0.0012743379,0.00009732813,0.0000020107802],"about_ca_topic_score_codex":0.004526214,"about_ca_topic_score_gemma":0.002491656,"teacher_disagreement_score":0.004526214,"about_ca_system_score_codex":0.0007989153,"about_ca_system_score_gemma":0.00084189273,"threshold_uncertainty_score":0.008999765},"labels":[],"label_agreement":null},{"id":"W2029155741","doi":"10.1109/siu.2010.5651543","title":"When does feedback not increase capacity for channels with memory?","year":2010,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Independent and identically distributed random variables; Markov process; Symmetry (geometry); Markov chain; Finite state; Topology (electrical circuits); Channel (broadcasting); Channel capacity; Control theory (sociology); Distributed computing; Algorithm; Theoretical computer science; Mathematics; Random variable; Computer network; Artificial intelligence; Machine learning; Combinatorics; Statistics","score_opus":0.01784659395454224,"score_gpt":0.22661355291017482,"score_spread":0.20876695895563258,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2029155741","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.55875504,0.001389247,0.40719655,0.008721784,0.00050294324,0.00016076902,0.00073812687,0.0015397476,0.020995729],"genre_scores_gemma":[0.99112767,0.0002005976,0.007042568,0.0002704987,0.000087637854,0.000051143576,0.00003805528,0.00006575325,0.0011160536],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9985306,0.00038618068,0.00006748667,0.00022727803,0.00025051346,0.00053785235],"domain_scores_gemma":[0.97174203,0.022060066,0.0020557565,0.0017845281,0.0012614022,0.0010963],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023528533,0.00053922,0.0013150363,0.0005681268,0.0008627068,0.0016685772,0.0015046871,0.0018104453,0.0068596285],"category_scores_gemma":[0.032903213,0.0003948228,0.00057183625,0.00042858435,0.00245893,0.0068349354,0.0016595899,0.0016618266,0.00048562544],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001863077,0.00044494777,0.004575525,0.0009653237,0.000180907,0.0009416549,0.0004357994,0.2941146,0.0314004,0.5118064,0.01191864,0.14135262],"study_design_scores_gemma":[0.00017355221,0.00041646234,0.0014038678,0.000114559916,0.000072874434,0.0004720005,0.000316551,0.54044867,0.0161265,0.4378108,0.002545895,0.00009825404],"about_ca_topic_score_codex":0.0014497738,"about_ca_topic_score_gemma":0.0012043385,"teacher_disagreement_score":0.0068596285,"about_ca_system_score_codex":0.001370937,"about_ca_system_score_gemma":0.0014077277,"threshold_uncertainty_score":0.022947729},"labels":[],"label_agreement":null},{"id":"W2034217237","doi":"10.1145/2668956.2668963","title":"On designing migrating agents","year":2014,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Human–computer interaction; Autonomous agent; Architecture; Distributed computing; Robot; Multi-agent system; Testbed; Avatar; Agent architecture; Artificial intelligence","score_opus":0.02156622164906063,"score_gpt":0.25015359849880964,"score_spread":0.228587376849749,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2034217237","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025276033,0.00052889384,0.96128154,0.00057211175,0.00011608643,0.00023127111,0.000023987865,0.00062581844,0.011344284],"genre_scores_gemma":[0.2524955,0.0010551705,0.7372439,0.00034451208,0.000041764095,0.00043946953,0.0001103811,0.00025339107,0.008015855],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9992231,0.0002671826,0.00006274123,0.00016888273,0.0001826223,0.000095430754],"domain_scores_gemma":[0.9989153,0.00033033293,0.00016835654,0.00028859611,0.00018870582,0.00010860839],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014319577,0.0008268475,0.00040931316,0.00035083087,0.0010404473,0.0013054133,0.0014682729,0.0011080903,0.002207995],"category_scores_gemma":[0.0044995183,0.0004887035,0.00059751986,0.00031057853,0.0016561672,0.0021062032,0.0026092054,0.0012175065,0.00064347277],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019597846,0.00021583335,0.0047593573,0.0010165881,0.00012043029,0.00074515323,0.0033066496,0.31258172,0.03948077,0.31729826,0.0054998524,0.3147794],"study_design_scores_gemma":[0.00013301306,0.00056733703,0.0011416539,0.00040239867,0.0001615631,0.0009463867,0.0013849034,0.5922651,0.028556032,0.21794802,0.15641724,0.000076369346],"about_ca_topic_score_codex":0.0012776733,"about_ca_topic_score_gemma":0.0018114125,"teacher_disagreement_score":0.002207995,"about_ca_system_score_codex":0.0005707317,"about_ca_system_score_gemma":0.001024404,"threshold_uncertainty_score":0.0075730085},"labels":[],"label_agreement":null},{"id":"W2034960258","doi":"10.1109/devlrn.2012.6400860","title":"Scaling life-long off-policy learning","year":2012,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Estimator; Artificial intelligence; Scaling; Convergence (economics); Machine learning; Scale (ratio); Value (mathematics); Coding (social sciences); Mathematics; Economics; Economic growth","score_opus":0.030943412808817627,"score_gpt":0.287813299354403,"score_spread":0.25686988654558535,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2034960258","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14274909,0.00051936094,0.85033256,0.0005506453,0.00017417729,0.00014563803,0.000057938763,0.0011661419,0.0043043382],"genre_scores_gemma":[0.902022,0.0002168457,0.09486202,0.00027198542,0.00007741109,0.0001925341,0.00011134054,0.0001922411,0.0020536198],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985251,0.00041841797,0.00009358373,0.00038804844,0.00041600404,0.00015883945],"domain_scores_gemma":[0.9888137,0.0060835835,0.001037075,0.0023433648,0.0011804908,0.0005417267],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037413961,0.0010663327,0.0012981639,0.000618497,0.0006432827,0.0014384586,0.0019880787,0.0011387711,0.002858651],"category_scores_gemma":[0.020076104,0.0005604941,0.00056740793,0.00044724467,0.0020449015,0.00398849,0.0021741984,0.0026011951,0.00053808436],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021611525,0.00027891612,0.002229507,0.00010411288,0.00007529949,0.000090238136,0.00014646522,0.90924394,0.004269886,0.022165105,0.0010346263,0.060145877],"study_design_scores_gemma":[0.000014530444,0.00008206074,0.00016787903,0.000006876275,0.0000070320416,0.000023064036,0.0000143843245,0.98283666,0.0013836187,0.014950009,0.00050538295,0.000008592091],"about_ca_topic_score_codex":0.0015887022,"about_ca_topic_score_gemma":0.0010830726,"teacher_disagreement_score":0.0037413961,"about_ca_system_score_codex":0.0014087362,"about_ca_system_score_gemma":0.0012483549,"threshold_uncertainty_score":0.019786656},"labels":[],"label_agreement":null},{"id":"W2038818712","doi":"10.1111/j.0824-7935.2004.00252.x","title":"The Advantages of Designing Adaptive Business Agents Using Reputation Modeling Compared to the Approach of Recursive Modeling","year":2004,"lang":"en","type":"article","venue":"Computational Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reputation; Computer science; Purchasing; Exploit; Reinforcement learning; Quality (philosophy); Risk analysis (engineering); Complex adaptive system; Artificial intelligence; Computer security; Business; Marketing","score_opus":0.1203924856587944,"score_gpt":0.32916524780149004,"score_spread":0.20877276214269563,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2038818712","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013625016,0.00006908922,0.98383266,0.00016032657,0.000010648255,0.000070275535,0.000007394805,0.00026636434,0.0019582626],"genre_scores_gemma":[0.61638665,0.00020845546,0.38095975,0.0001348984,0.000032915952,0.00023881855,0.000045472727,0.000064905136,0.0019281929],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99874854,0.00061960577,0.00007453476,0.00024103245,0.00022525156,0.000091010996],"domain_scores_gemma":[0.9968078,0.0016389847,0.00043176077,0.0006857577,0.00029344446,0.00014238658],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017618379,0.0007690507,0.00081340806,0.00031244717,0.00047088455,0.0011260214,0.0015103316,0.0013411051,0.0013724446],"category_scores_gemma":[0.0067723133,0.00044997293,0.00070295786,0.00031948948,0.0010087187,0.0023277025,0.0009788264,0.0011156847,0.00067137886],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011221887,0.00023949912,0.0028115062,0.00018694665,0.00015890608,0.00015878322,0.00039962493,0.78004885,0.013176612,0.086198375,0.00082453823,0.11568418],"study_design_scores_gemma":[0.000028547354,0.00008496939,0.00016779028,0.0000084346,0.00002504687,0.00006121656,0.000020089998,0.98461676,0.0017610518,0.011658106,0.0015516889,0.000016240976],"about_ca_topic_score_codex":0.0018847504,"about_ca_topic_score_gemma":0.0023056944,"teacher_disagreement_score":0.0018847504,"about_ca_system_score_codex":0.0005650828,"about_ca_system_score_gemma":0.001131745,"threshold_uncertainty_score":0.009317577},"labels":[],"label_agreement":null},{"id":"W2045192098","doi":"10.1109/iat.2006.108","title":"Resolution-Based Policy Search for Imperfect Information Differential Games","year":2006,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Perfect information; Imperfect; Computer science; Discretization; Differential game; Sequential game; Differential (mechanical device); Mathematical optimization; Zero-sum game; Game theory; Nash equilibrium; Mathematical economics; Mathematics","score_opus":0.0111374520453231,"score_gpt":0.2576978047175407,"score_spread":0.24656035267221763,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2045192098","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034196444,0.00020523753,0.96325296,0.00021239076,0.000027185964,0.00005343678,0.000021407526,0.00009604257,0.0019348435],"genre_scores_gemma":[0.8583748,0.00017759071,0.13951766,0.00011591776,0.000026726153,0.00019509936,0.00005265094,0.000033782893,0.0015058098],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99899286,0.00061464764,0.000044932378,0.000115727795,0.00014306307,0.000088821915],"domain_scores_gemma":[0.99413496,0.004709482,0.00051469466,0.00021567308,0.000197699,0.00022755927],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025003394,0.000757799,0.0015325237,0.00066448,0.00040729422,0.00086292555,0.0013221942,0.0011861104,0.0012433034],"category_scores_gemma":[0.009940548,0.0005876687,0.00058345916,0.00044974097,0.0015921193,0.0013141269,0.0018600302,0.0015255125,0.00015721904],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000040615465,0.000026600048,0.00027053556,0.000035913745,0.000022178992,0.00004636437,0.000047321908,0.97034025,0.00033569144,0.021258913,0.00019706841,0.0073786955],"study_design_scores_gemma":[0.000010697379,0.000010722721,0.000017640445,0.0000027245726,0.0000018241152,0.0000057687844,0.0000045589727,0.99449265,0.00008258805,0.0052782306,0.00009036832,0.0000022162387],"about_ca_topic_score_codex":0.0020395536,"about_ca_topic_score_gemma":0.0012147536,"teacher_disagreement_score":0.0025003394,"about_ca_system_score_codex":0.0010162871,"about_ca_system_score_gemma":0.00092848256,"threshold_uncertainty_score":0.013223231},"labels":[],"label_agreement":null},{"id":"W2046578831","doi":"10.1109/iros.2005.1545097","title":"Active versus passive expression of preference in the control of multiple-robot decision-making","year":2005,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Preference; Robot; Expression (computer science); Computer science; Mechanism (biology); Control (management); Quality (philosophy); Artificial intelligence; Human–computer interaction; Mathematics","score_opus":0.033262968542642296,"score_gpt":0.28840695881588035,"score_spread":0.2551439902732381,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2046578831","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46030146,0.00045926648,0.5330225,0.00038759943,0.00005665597,0.00009360439,0.000018948707,0.0002592832,0.0054007038],"genre_scores_gemma":[0.9632678,0.00006572668,0.035824288,0.000050634735,0.000012276275,0.000049650633,0.000005417959,0.000014650213,0.00070958346],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99865013,0.0007068577,0.000075401316,0.0002051756,0.0002296022,0.00013289307],"domain_scores_gemma":[0.9958352,0.002349952,0.0007387157,0.0005491423,0.00023872213,0.00028819893],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026916105,0.00046889335,0.0003980129,0.00019813167,0.00027913562,0.00095482543,0.000995138,0.0004966654,0.0010106032],"category_scores_gemma":[0.0057179313,0.00019109392,0.00037796458,0.00022840909,0.0018038917,0.0012082654,0.0009279051,0.0010607606,0.000115800685],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017753382,0.0007495206,0.008134178,0.00061750447,0.00034958843,0.00036585756,0.0014658173,0.35434192,0.21357766,0.19615303,0.00062601984,0.2218436],"study_design_scores_gemma":[0.00034816485,0.0011108327,0.002830258,0.000034004024,0.00011108142,0.00018651257,0.00015907688,0.84296864,0.055308316,0.09424834,0.0025924442,0.000102304504],"about_ca_topic_score_codex":0.0004503864,"about_ca_topic_score_gemma":0.0006071803,"teacher_disagreement_score":0.0026916105,"about_ca_system_score_codex":0.00048665397,"about_ca_system_score_gemma":0.0004481178,"threshold_uncertainty_score":0.014234781},"labels":[],"label_agreement":null},{"id":"W2047918528","doi":"10.1145/1329125.1329174","title":"Multiagent learning in adaptive dynamic systems","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Class (philosophy); Set (abstract data type); Interdependence; Reinforcement learning; Artificial intelligence; Exploit; Adaptive learning; Mathematical optimization; Mathematics","score_opus":0.015793344502379895,"score_gpt":0.2588840668079719,"score_spread":0.24309072230559198,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2047918528","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009011366,0.005214342,0.97584116,0.0013049773,0.00020360123,0.00007860285,0.000040415,0.00016726999,0.008138277],"genre_scores_gemma":[0.8229545,0.0056807683,0.16083881,0.00047605802,0.00078331,0.0006600549,0.00012581352,0.000056934838,0.008423718],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988605,0.00056968926,0.00006933817,0.00020071055,0.00022583803,0.00007395357],"domain_scores_gemma":[0.99791104,0.001567385,0.0001586608,0.00010203906,0.00018478879,0.00007603414],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016752526,0.0011450988,0.001358286,0.0005511309,0.0005834461,0.0016200136,0.0012251489,0.002001234,0.0024117772],"category_scores_gemma":[0.004078837,0.0004036699,0.0006361351,0.0007662895,0.0019509085,0.0014229444,0.0015045925,0.0020417026,0.0004393849],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000044767534,0.00006500506,0.0007811048,0.0002383195,0.000099307654,0.00016582599,0.00014646,0.7056974,0.00056184834,0.25106096,0.0015731977,0.0395658],"study_design_scores_gemma":[0.000024495057,0.000039952338,0.0001269673,0.000019926802,0.000010658097,0.000023610419,0.00002034734,0.8766165,0.00018719169,0.119952396,0.0029674927,0.0000104904975],"about_ca_topic_score_codex":0.0035400214,"about_ca_topic_score_gemma":0.0018511446,"teacher_disagreement_score":0.0035400214,"about_ca_system_score_codex":0.0013486242,"about_ca_system_score_gemma":0.0009787547,"threshold_uncertainty_score":0.009785056},"labels":[],"label_agreement":null},{"id":"W2050992779","doi":"10.1103/physreve.67.026706","title":"Convergence of reinforcement learning algorithms and acceleration of learning","year":2003,"lang":"en","type":"article","venue":"Physical review. E, Statistical physics, plasmas, fluids, and related interdisciplinary topics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Acceleration; Reinforcement learning; Convergence (economics); Rate of convergence; Computer science; Conjecture; Algorithm; Relation (database); Popularity; Reinforcement; Mathematical optimization; Mathematics; Applied mathematics; Artificial intelligence; Physics; Discrete mathematics; Quantum mechanics","score_opus":0.01570648697182246,"score_gpt":0.3051240425009342,"score_spread":0.28941755552911175,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2050992779","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013734327,0.0015202754,0.9766406,0.0004946393,0.0001195031,0.00008456391,0.00002621122,0.00038466524,0.0069953175],"genre_scores_gemma":[0.7347751,0.0026290438,0.25224248,0.00038045878,0.0003244339,0.0006739852,0.00012847778,0.00023338877,0.008612641],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998767,0.00051395665,0.000066729706,0.00020640233,0.000300362,0.00014563506],"domain_scores_gemma":[0.99357516,0.004744235,0.00047239207,0.00035487686,0.0006819444,0.00017142127],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003220737,0.0013821418,0.0014095723,0.001006778,0.00058295194,0.0012383597,0.0014490216,0.0017411286,0.004074197],"category_scores_gemma":[0.017523164,0.0005652005,0.0006980032,0.00080123416,0.0020983976,0.001731659,0.0018590284,0.0026467182,0.0008208487],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011594392,0.00008286572,0.0008880911,0.00019543689,0.00006717682,0.00004206786,0.00009887942,0.84140426,0.0010004855,0.08912071,0.0013747418,0.065609336],"study_design_scores_gemma":[0.000024527066,0.000053330154,0.000097291035,0.000027276436,0.000007276404,0.00001617682,0.000007562822,0.9597124,0.00035028148,0.03886237,0.0008339635,0.0000074491363],"about_ca_topic_score_codex":0.0033719812,"about_ca_topic_score_gemma":0.0016618298,"teacher_disagreement_score":0.004074197,"about_ca_system_score_codex":0.0012922309,"about_ca_system_score_gemma":0.0013303009,"threshold_uncertainty_score":0.0170331},"labels":[],"label_agreement":null},{"id":"W2051178751","doi":"10.1145/1160633.1160764","title":"Learning the required number of agents for complex tasks","year":2006,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Task (project management); Computer science; Reinforcement learning; Perception; Artificial intelligence; Human–computer interaction; Machine learning; Psychology; Engineering","score_opus":0.044996304994496285,"score_gpt":0.3049986134090903,"score_spread":0.260002308414594,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2051178751","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25672948,0.000119036195,0.7391304,0.00033854455,0.000041407788,0.00015011281,0.00009223888,0.0009719302,0.0024268185],"genre_scores_gemma":[0.8112982,0.00009519452,0.18650623,0.000058855516,0.00002105747,0.00020165945,0.00012877498,0.00011175484,0.0015783657],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99936134,0.00013913594,0.00006582195,0.00020708251,0.00012114224,0.00010542278],"domain_scores_gemma":[0.99435365,0.0028085073,0.00081593526,0.0008793045,0.0006500425,0.0004925496],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015217048,0.00081331155,0.00093335024,0.00023177014,0.0004227363,0.00075025146,0.0015542692,0.00087294466,0.0026384788],"category_scores_gemma":[0.009985969,0.0005366144,0.00040741387,0.00018707807,0.0006461531,0.0022258563,0.001048313,0.0015360833,0.00048788902],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006263649,0.00051528896,0.0041247504,0.00023320968,0.000055351706,0.00010126626,0.00015466212,0.8409966,0.019574035,0.008240524,0.0010445111,0.12433346],"study_design_scores_gemma":[0.00006678099,0.0001238968,0.00062893983,0.000008372525,0.000009928617,0.000036530928,0.000027012005,0.987252,0.0055667628,0.005861391,0.0004080551,0.0000103759485],"about_ca_topic_score_codex":0.0015953287,"about_ca_topic_score_gemma":0.002378384,"teacher_disagreement_score":0.0026384788,"about_ca_system_score_codex":0.0008086278,"about_ca_system_score_gemma":0.0016876005,"threshold_uncertainty_score":0.008826613},"labels":[],"label_agreement":null},{"id":"W2055622634","doi":"10.1109/cig.2009.5286455","title":"Improving testing of multi-unit computer players for unwanted behavior using coordination macros","year":2009,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Macro; Computer science; Unit testing; Competition (biology); Strengths and weaknesses; Limit (mathematics); Resource (disambiguation); Software; Operating system; Programming language","score_opus":0.07838992345998244,"score_gpt":0.3182366372027885,"score_spread":0.23984671374280606,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2055622634","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27365845,0.00006824756,0.7179497,0.00014699427,0.000040064537,0.00031107027,0.00006516738,0.0053657982,0.0023945784],"genre_scores_gemma":[0.76676404,0.00003160483,0.23137021,0.00008072385,0.000009427171,0.0002064533,0.00007477493,0.00020186247,0.0012609597],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974533,0.0008491152,0.000201294,0.00050757243,0.00079863204,0.00018999084],"domain_scores_gemma":[0.98984855,0.0054813162,0.0011646986,0.0017878521,0.0011889633,0.0005286824],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024254334,0.00095658505,0.000662299,0.00058075663,0.00029693075,0.0007377793,0.0022146031,0.0006562502,0.0021701904],"category_scores_gemma":[0.010933021,0.000414452,0.00041194723,0.00020737253,0.0011041488,0.0013661354,0.0014173451,0.0010071875,0.0004459913],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014518505,0.0019328444,0.030279476,0.000423671,0.00022635126,0.00096299016,0.0009973258,0.22157969,0.29200625,0.013138355,0.0017239244,0.43527722],"study_design_scores_gemma":[0.000081713784,0.0010329144,0.003133103,0.000019388264,0.000036975336,0.00027038544,0.000056346256,0.9141359,0.07660203,0.0032031746,0.0013785706,0.000049463182],"about_ca_topic_score_codex":0.0019852908,"about_ca_topic_score_gemma":0.0016074551,"teacher_disagreement_score":0.0024254334,"about_ca_system_score_codex":0.0005138659,"about_ca_system_score_gemma":0.00095685845,"threshold_uncertainty_score":0.012827039},"labels":[],"label_agreement":null},{"id":"W2055921164","doi":"10.1007/s10458-012-9200-2","title":"A survey of point-based POMDP solvers","year":2012,"lang":"en","type":"article","venue":"Autonomous Agents and Multi-Agent Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":433,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Partially observable Markov decision process; Markov decision process; Computer science; Bellman equation; Function (biology); Observable; Value (mathematics); Point (geometry); State space; Mathematical optimization; Theoretical computer science; Space (punctuation); Scale (ratio); Markov chain; Markov process; Mathematics; Markov model; Machine learning","score_opus":0.06430470677447479,"score_gpt":0.28584377584545256,"score_spread":0.22153906907097776,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2055921164","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0019540193,0.008520793,0.9814387,0.00026697555,0.00010075221,0.00008090495,0.00013235926,0.00068773446,0.0068178372],"genre_scores_gemma":[0.118885204,0.024617137,0.84826297,0.00041781476,0.00034379357,0.00056700903,0.00088466314,0.00047175607,0.0055496343],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989374,0.00029575298,0.00011221246,0.0001586496,0.00042897576,0.000067078014],"domain_scores_gemma":[0.9987143,0.0008001769,0.00006625631,0.00014506267,0.00023244595,0.000041753232],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015685823,0.001619746,0.002730085,0.0012371318,0.0006101618,0.002119256,0.0040091393,0.0018575037,0.006367066],"category_scores_gemma":[0.0040429784,0.0011555156,0.0015403294,0.003400004,0.0008185172,0.0020589572,0.002403722,0.0022292084,0.0024083573],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010457624,0.00011788833,0.00054641813,0.0012383085,0.00016622432,0.000050988558,0.000071665985,0.52662337,0.0006656771,0.05297909,0.0076984004,0.4097374],"study_design_scores_gemma":[0.000055487122,0.000048419024,0.00012639757,0.00014830244,0.000041359137,0.00004812752,0.000024196921,0.950703,0.0005168721,0.036987953,0.011283031,0.000016816532],"about_ca_topic_score_codex":0.004555707,"about_ca_topic_score_gemma":0.004043397,"teacher_disagreement_score":0.006367066,"about_ca_system_score_codex":0.0008268436,"about_ca_system_score_gemma":0.0016999706,"threshold_uncertainty_score":0.021299958},"labels":[],"label_agreement":null},{"id":"W2060652090","doi":"10.1142/s0129183104006662","title":"REINFORCEMENT LEARNING WITH GOAL-DIRECTED ELIGIBILITY TRACES","year":2004,"lang":"en","type":"article","venue":"International Journal of Modern Physics C","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Reinforcement learning; TRACE (psycholinguistics); Computer science; Reinforcement; Artificial intelligence; Mechanism (biology); Machine learning; Psychology","score_opus":0.014142522122078973,"score_gpt":0.27601330258619006,"score_spread":0.2618707804641111,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2060652090","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022279708,0.00006244262,0.97563,0.00012390257,0.000028317676,0.000046657817,0.000021301432,0.00039434442,0.0014133691],"genre_scores_gemma":[0.8484276,0.0001531558,0.1480691,0.00012756875,0.000037153735,0.00017739447,0.0000858792,0.00007964038,0.0028425131],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99940896,0.00020436019,0.000037347527,0.000093870716,0.000185306,0.000070300965],"domain_scores_gemma":[0.99813664,0.0010792618,0.00014829742,0.00021993826,0.0002573132,0.00015851486],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010500371,0.00053355476,0.0005816521,0.000291762,0.00028889385,0.0006142298,0.0013442386,0.0007710444,0.001632248],"category_scores_gemma":[0.006544967,0.00020773211,0.000327703,0.0002899247,0.0008837685,0.0016083408,0.0012368265,0.0015267739,0.00023354945],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042594576,0.00030077156,0.0011211078,0.00017096645,0.00007163097,0.00028615264,0.00017155803,0.5883097,0.016386125,0.2356854,0.0018952988,0.15517536],"study_design_scores_gemma":[0.000058383772,0.00007667146,0.000093920484,0.0000061846094,0.000007880059,0.000030646257,0.000006111112,0.9374064,0.002802035,0.058729555,0.00077242334,0.000009798534],"about_ca_topic_score_codex":0.0013390535,"about_ca_topic_score_gemma":0.0012877696,"teacher_disagreement_score":0.001632248,"about_ca_system_score_codex":0.0004325515,"about_ca_system_score_gemma":0.0012474165,"threshold_uncertainty_score":0.005553186},"labels":[],"label_agreement":null},{"id":"W2064203730","doi":"10.5555/777092.777139","title":"Greedy linear value-approximation for factored Markov decision processes","year":2002,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; University of Waterloo","funders":"","keywords":"Markov decision process; Bellman equation; Mathematical optimization; Linear programming; Computer science; Dynamic programming; Basis (linear algebra); Set (abstract data type); Linear approximation; Approximation error; Approximation algorithm; Function (biology); Value (mathematics); Function approximation; Markov process; Mathematics; Artificial intelligence; Nonlinear system; Artificial neural network","score_opus":0.03629519433379755,"score_gpt":0.266278795498935,"score_spread":0.22998360116513747,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2064203730","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0077197314,0.00032758972,0.98985785,0.00016610925,0.000020180649,0.000039673738,0.000047219393,0.00028236335,0.0015393121],"genre_scores_gemma":[0.59311277,0.00069359917,0.40156835,0.0001964227,0.00006344167,0.00044904012,0.0002869368,0.00019561045,0.0034339302],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9981262,0.00093304046,0.000079278485,0.0002868158,0.00034829052,0.00022643885],"domain_scores_gemma":[0.9919477,0.006787943,0.0004384128,0.00030087263,0.00035241622,0.00017265276],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037974936,0.0015233387,0.002399641,0.00097313576,0.00058596785,0.0014092596,0.0015239168,0.0017309298,0.0035199127],"category_scores_gemma":[0.013363992,0.0009371208,0.0010431673,0.0011332774,0.0019247596,0.001764017,0.001744977,0.0026018808,0.0006064524],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005090429,0.000021123875,0.0002020885,0.000048177295,0.000019604777,0.000026181102,0.000034291752,0.96491855,0.00014523289,0.023446484,0.00043163143,0.0106557775],"study_design_scores_gemma":[0.000006788144,0.000007705279,0.000013225086,0.000005902588,0.0000023617388,0.0000033002862,0.00000259908,0.98563933,0.00005714297,0.014152654,0.00010709914,0.000001985933],"about_ca_topic_score_codex":0.009016861,"about_ca_topic_score_gemma":0.0074165515,"teacher_disagreement_score":0.009016861,"about_ca_system_score_codex":0.003226116,"about_ca_system_score_gemma":0.0026169652,"threshold_uncertainty_score":0.02340722},"labels":[],"label_agreement":null},{"id":"W2066471519","doi":"10.3166/ria.21.9-33","title":"Apprentissage actif dans les processus décisionnels de Markov partiellement observables L'algorithme MEDUSA","year":2007,"lang":"fr","type":"article","venue":"Revue d intelligence artificielle","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Physics; Philosophy","score_opus":0.07443563189771293,"score_gpt":0.3055678652788824,"score_spread":0.23113223338116948,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2066471519","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041059174,0.00016672697,0.95613235,0.0003492285,0.00003494947,0.000052680003,0.000032243388,0.0006952955,0.0014773079],"genre_scores_gemma":[0.63375944,0.00022932792,0.36016095,0.0002498233,0.000057191803,0.00026286562,0.000144093,0.0001359375,0.0050003896],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986325,0.0005101023,0.00005257005,0.00032685738,0.00026263925,0.00021529908],"domain_scores_gemma":[0.9961505,0.0027546778,0.00021210365,0.0002842866,0.00035971447,0.00023879818],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023845183,0.0011172551,0.0016555641,0.0005799911,0.000895172,0.0019497307,0.0015777653,0.0019707528,0.002471478],"category_scores_gemma":[0.0075544156,0.00069015427,0.001136793,0.00048754076,0.0023315162,0.002138299,0.0019790744,0.0025641564,0.00046129487],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004069236,0.000120835386,0.001168531,0.00011736343,0.00007009695,0.00015364526,0.000266407,0.88590246,0.0026232766,0.054444425,0.0010965049,0.05362953],"study_design_scores_gemma":[0.00003020308,0.000044734894,0.00004742046,0.0000067522005,0.0000054399748,0.000014609699,0.000014830614,0.9898122,0.0008078468,0.008776938,0.0004332137,0.0000057590237],"about_ca_topic_score_codex":0.007393787,"about_ca_topic_score_gemma":0.007236946,"teacher_disagreement_score":0.007393787,"about_ca_system_score_codex":0.0014773182,"about_ca_system_score_gemma":0.0023331873,"threshold_uncertainty_score":0.014701486},"labels":[],"label_agreement":null},{"id":"W2067564749","doi":"10.1142/s0218213012500030","title":"STOCHASTIC RESOURCE ALLOCATION IN MULTIAGENT ENVIRONMENTS: AN APPROACH BASED ON DISTRIBUTED Q-VALUES AND BOUNDED REAL-TIME DYNAMIC PROGRAMMING","year":2012,"lang":"en","type":"article","venue":"International Journal of Artificial Intelligence Tools","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"Fonds Québécois de la Recherche sur la Nature et les Technologies","keywords":"Computer science; Bounded function; A priori and a posteriori; Mathematical optimization; Heuristic; Convergence (economics); Set (abstract data type); Multi-agent system; Distributed computing; Resource allocation; Dynamic programming; Distributed algorithm; Resource (disambiguation); Reduction (mathematics); Algorithm; Artificial intelligence; Mathematics","score_opus":0.03907911182316321,"score_gpt":0.3112756167984046,"score_spread":0.2721965049752414,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2067564749","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0017194379,0.00009291098,0.99733436,0.000093230155,0.000018094099,0.000021587839,0.0000048975235,0.00003476541,0.0006807503],"genre_scores_gemma":[0.46519917,0.000545579,0.53073776,0.00024154341,0.00014280516,0.00043989607,0.000049523056,0.00013031435,0.0025134382],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99787414,0.0011140166,0.000080932296,0.00029515178,0.0004605541,0.00017521618],"domain_scores_gemma":[0.9971265,0.0020197327,0.00026422943,0.00016523675,0.0002498079,0.00017438736],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003010325,0.0011808224,0.00184924,0.00077279226,0.00063672836,0.0016358419,0.0024673855,0.0013759158,0.0019347547],"category_scores_gemma":[0.005009709,0.00078711257,0.0011899638,0.0009228187,0.0017901942,0.001979548,0.0023409508,0.0020283533,0.0002823187],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000034953737,0.0000504555,0.00017401099,0.00007048296,0.000040533665,0.00007299742,0.000078084806,0.91807616,0.00065681484,0.06084831,0.00044295596,0.019454168],"study_design_scores_gemma":[0.00001061516,0.000021432112,0.00002433336,0.0000058589007,0.0000058010114,0.000012965618,0.000007855568,0.97444916,0.00014415548,0.024823852,0.00048795316,0.0000059400513],"about_ca_topic_score_codex":0.0022721663,"about_ca_topic_score_gemma":0.0019737442,"teacher_disagreement_score":0.003010325,"about_ca_system_score_codex":0.001368394,"about_ca_system_score_gemma":0.0020100244,"threshold_uncertainty_score":0.015920281},"labels":[],"label_agreement":null},{"id":"W2069891133","doi":"10.1109/ciisp.2007.369176","title":"Application of Opposition-Based Reinforcement Learning in Image Segmentation","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Image segmentation; Artificial intelligence; Segmentation; Exploit; Computer vision; Image texture; Pattern recognition (psychology); Computer security","score_opus":0.009708222293382411,"score_gpt":0.26751706800371006,"score_spread":0.25780884571032764,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2069891133","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013992233,0.00017165237,0.98303,0.00015698184,0.000032620756,0.00005120438,0.0000052692185,0.00017562883,0.0023843893],"genre_scores_gemma":[0.7870074,0.00018553383,0.20990653,0.00017869742,0.00004150853,0.00018501138,0.000019770354,0.000054045715,0.0024215213],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991856,0.0004160542,0.000029896793,0.00009695064,0.00021817264,0.000053309508],"domain_scores_gemma":[0.99808633,0.0013483944,0.00017289778,0.00009517122,0.0002171538,0.0000801234],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016065569,0.00054595293,0.0009078321,0.00044677124,0.00033330132,0.00059429836,0.0010175247,0.0009092725,0.0013049763],"category_scores_gemma":[0.0038901507,0.00027791754,0.00044252377,0.00030674596,0.0013276297,0.0006468817,0.00093652646,0.000938403,0.00019544161],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012921359,0.000121578756,0.0006232925,0.00010219151,0.00006757964,0.00015672763,0.00012543902,0.9026889,0.0066242334,0.02298944,0.0005167844,0.06585458],"study_design_scores_gemma":[0.00001735581,0.00006767957,0.000065304484,0.0000063140724,0.0000050459275,0.000029323106,0.000005064827,0.9932963,0.00092300883,0.005068739,0.00050912343,0.000006750002],"about_ca_topic_score_codex":0.001681329,"about_ca_topic_score_gemma":0.0011676096,"teacher_disagreement_score":0.001681329,"about_ca_system_score_codex":0.0007464845,"about_ca_system_score_gemma":0.00069080904,"threshold_uncertainty_score":0.008496344},"labels":[],"label_agreement":null},{"id":"W2071814471","doi":"10.1145/1102351.1102472","title":"Bayesian sparse sampling for on-line reward optimization","year":2005,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":122,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Machine learning; Bayesian probability; Sampling (signal processing); Action selection; Artificial intelligence; Bayesian optimization; Thompson sampling; Selection (genetic algorithm); Bayesian inference; Mathematical optimization; Mathematics","score_opus":0.06390024574608347,"score_gpt":0.30902289421512674,"score_spread":0.24512264846904327,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2071814471","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0039453115,0.00012958226,0.9946301,0.00011740243,0.000015765872,0.00002922191,0.000026834718,0.00018899003,0.000916827],"genre_scores_gemma":[0.5552104,0.00039453543,0.43991885,0.00030176804,0.00011190033,0.00045781923,0.0002412844,0.00018299806,0.003180459],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984425,0.0007559828,0.00004584587,0.00013932584,0.000466181,0.00015023715],"domain_scores_gemma":[0.995276,0.003629533,0.00025942197,0.00027243252,0.00040765209,0.00015484782],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002255764,0.0009223177,0.0016443946,0.0007098412,0.00047460973,0.00094503054,0.0014651701,0.0013173104,0.003496309],"category_scores_gemma":[0.012440967,0.0007488164,0.00054424256,0.0008439115,0.0012233194,0.0016037486,0.0013865029,0.0021442692,0.0007544998],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010665788,0.00007068943,0.0003730438,0.00006610405,0.000025353793,0.000035499244,0.000040852876,0.92089564,0.0007691351,0.0339901,0.0015332355,0.042093802],"study_design_scores_gemma":[0.000008057229,0.000008654721,0.000019938963,0.000003585806,0.0000017671892,0.000003524868,0.0000013769632,0.99007636,0.0001279549,0.009598591,0.00014840737,0.0000018847799],"about_ca_topic_score_codex":0.0045991475,"about_ca_topic_score_gemma":0.0058154343,"teacher_disagreement_score":0.0045991475,"about_ca_system_score_codex":0.0013550596,"about_ca_system_score_gemma":0.0015822554,"threshold_uncertainty_score":0.01192981},"labels":[],"label_agreement":null},{"id":"W2073549596","doi":"10.1109/icsmc.2012.6378016","title":"Acquiring a broad range of empirical knowledge in real time by temporal-difference learning","year":2012,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Generality; Robot; Computer science; Mobile robot; Set (abstract data type); Range (aeronautics); Artificial intelligence; Interface (matter); Robot learning; Human–computer interaction; Engineering","score_opus":0.036091105780360115,"score_gpt":0.31493739586074165,"score_spread":0.27884629008038153,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2073549596","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029752232,0.00011238664,0.9684419,0.00022616738,0.000016563345,0.000020154466,0.000052985113,0.0003151288,0.0010624649],"genre_scores_gemma":[0.7824989,0.00020786094,0.2155788,0.0001768174,0.000040020474,0.00009539147,0.00021969088,0.000059336988,0.0011231253],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99918133,0.00022741874,0.00006259678,0.00025195366,0.00023043284,0.000046134104],"domain_scores_gemma":[0.993232,0.004766849,0.00047879343,0.0010193901,0.0003827083,0.000120245946],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022032221,0.00067881285,0.00071015686,0.0005242245,0.0003201276,0.00087676494,0.001749011,0.0008483871,0.001485687],"category_scores_gemma":[0.012530681,0.00048053465,0.0006283356,0.0005992728,0.0015798678,0.0031401287,0.0012237483,0.0018052341,0.00029221716],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030337143,0.00027001742,0.0039583966,0.00015989707,0.0001299364,0.00018185393,0.00022524233,0.69549227,0.0121194115,0.04456473,0.0012889658,0.24130584],"study_design_scores_gemma":[0.00001183093,0.000032409007,0.00048803617,0.0000088258475,0.0000059830745,0.000030482952,0.000012175384,0.9713274,0.0021869764,0.025518037,0.00036660198,0.000011174039],"about_ca_topic_score_codex":0.0026962622,"about_ca_topic_score_gemma":0.002938096,"teacher_disagreement_score":0.0026962622,"about_ca_system_score_codex":0.0010256408,"about_ca_system_score_gemma":0.00071967114,"threshold_uncertainty_score":0.011651933},"labels":[],"label_agreement":null},{"id":"W2074304756","doi":"10.1109/ssrr.2013.6719367","title":"Learning to cooperate together: A semi-autonomous control architecture for multi-robot teams in urban search and rescue","year":2013,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Urban search and rescue; Search and rescue; Rescue robot; Robot; Reinforcement learning; Task (project management); Computer science; Identification (biology); Architecture; Human–computer interaction; Control (management); Collision avoidance; Artificial intelligence; Mobile robot; Computer security; Engineering; Collision; Systems engineering","score_opus":0.01678453592790917,"score_gpt":0.263294230890312,"score_spread":0.24650969496240283,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2074304756","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049838662,0.00010749523,0.94655746,0.00018673878,0.000029743702,0.000077093464,0.000010823716,0.00054621644,0.002645818],"genre_scores_gemma":[0.95014954,0.000043893728,0.0485854,0.000058067948,0.00001461775,0.00011590061,0.000016075599,0.0000116099345,0.0010049316],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996271,0.00011578404,0.000019925918,0.000084244486,0.00009725895,0.000055726723],"domain_scores_gemma":[0.99953485,0.00012229095,0.000098813754,0.000059709193,0.000121482146,0.0000628407],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007814268,0.00034904806,0.00031712942,0.00014742487,0.0003661355,0.00052427244,0.0010341421,0.00044219423,0.000837293],"category_scores_gemma":[0.00096851646,0.00016979888,0.00027689952,0.0001103697,0.00081162737,0.0004314213,0.0007512759,0.00054822344,0.00020109983],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000112009555,0.00016008373,0.0010970961,0.000075951386,0.000051648243,0.00018210565,0.00037144328,0.88715774,0.02416138,0.008679266,0.00070144376,0.077249914],"study_design_scores_gemma":[0.000017778351,0.00009842162,0.00020035123,0.0000037482082,0.000005841665,0.000020300376,0.000014532438,0.99657416,0.0011854676,0.0014650589,0.00040889444,0.0000055731616],"about_ca_topic_score_codex":0.0030682932,"about_ca_topic_score_gemma":0.002437746,"teacher_disagreement_score":0.0030682932,"about_ca_system_score_codex":0.00047737427,"about_ca_system_score_gemma":0.00085607974,"threshold_uncertainty_score":0.006100893},"labels":[],"label_agreement":null},{"id":"W2074458475","doi":"10.1016/j.engappai.2007.05.006","title":"A machine-learning approach to multi-robot coordination","year":2007,"lang":"en","type":"article","venue":"Engineering Applications of Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":81,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Reinforcement learning; Robot; Probabilistic logic; Scheme (mathematics); Task (project management); Object (grammar); Artificial intelligence; Robot learning; Genetic algorithm; Distributed computing; Mobile robot; Machine learning","score_opus":0.03084198381141881,"score_gpt":0.280624133256297,"score_spread":0.24978214944487817,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2074458475","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018644596,0.00031895653,0.99383014,0.00030743773,0.00006947234,0.000013967926,0.000010840452,0.000066073095,0.0035186014],"genre_scores_gemma":[0.5486359,0.0010787951,0.43955594,0.0003262521,0.00045463513,0.00025621548,0.000062508756,0.000101646416,0.009528113],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99940443,0.00025950422,0.000031208518,0.000104182174,0.00016214198,0.000038514187],"domain_scores_gemma":[0.9989317,0.0006851169,0.00009886004,0.00009431366,0.00013660964,0.00005344555],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009340197,0.0006108656,0.0010656086,0.0006987363,0.00060261425,0.0011782281,0.0022464648,0.0015784438,0.00317721],"category_scores_gemma":[0.00311861,0.00039807436,0.00069579616,0.000889715,0.0019125785,0.0014997828,0.0012816753,0.0014924805,0.00048603877],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000027327926,0.000036353427,0.00023058392,0.000084203566,0.000058034144,0.000092539514,0.00008720243,0.6617565,0.0010251184,0.29783612,0.0012202531,0.03754579],"study_design_scores_gemma":[0.000008030958,0.00001556502,0.00004599456,0.000006980679,0.0000061343503,0.000019027244,0.0000069951197,0.8822493,0.00016396261,0.11620313,0.0012682265,0.00000665948],"about_ca_topic_score_codex":0.0022881255,"about_ca_topic_score_gemma":0.0018895952,"teacher_disagreement_score":0.00317721,"about_ca_system_score_codex":0.000859278,"about_ca_system_score_gemma":0.0007239197,"threshold_uncertainty_score":0.01062876},"labels":[],"label_agreement":null},{"id":"W2077559824","doi":"10.1287/opre.1090.0705","title":"Acceleration Operators in the Value Iteration Algorithms for Markov Decision Processes","year":2009,"lang":"en","type":"article","venue":"Operations Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"GAVI Alliance","keywords":"Markov decision process; Monotone polygon; Mathematical optimization; Algorithm; Convergence (economics); Markov chain; Operator (biology); Mathematics; Computer science; Dynamic programming; Contraction (grammar); Acceleration; Linear programming; Markov process","score_opus":0.10822638641660903,"score_gpt":0.42880517446279043,"score_spread":0.3205787880461814,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2077559824","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006001553,0.00025498486,0.9922057,0.00013654683,0.000045603792,0.0000332683,0.00000590451,0.000063115585,0.001253296],"genre_scores_gemma":[0.3414166,0.0011004397,0.6517525,0.00021566972,0.00019262795,0.00045861906,0.000053112355,0.00013644408,0.004674017],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9987244,0.00066957145,0.0000560521,0.00014271369,0.0003073507,0.00009981072],"domain_scores_gemma":[0.99562114,0.0035197218,0.00021016979,0.00020057352,0.00032205618,0.00012638376],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036529468,0.0011358557,0.0010590042,0.00057971803,0.0004692837,0.0009886138,0.0012348372,0.0012116688,0.002225137],"category_scores_gemma":[0.011098422,0.0004779153,0.0010047241,0.0007437366,0.002026681,0.0020432041,0.0018788207,0.0027753266,0.00044971093],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013541894,0.000105528765,0.00054429757,0.00022124853,0.000054009397,0.00009641162,0.0002727772,0.44605622,0.0032086617,0.46303567,0.0013759681,0.08489369],"study_design_scores_gemma":[0.000017222292,0.00006335796,0.000032698685,0.000016619206,0.0000063434113,0.000023460363,0.000008085284,0.9199017,0.00076254044,0.07810619,0.0010529794,0.000008685378],"about_ca_topic_score_codex":0.0011182661,"about_ca_topic_score_gemma":0.00064889254,"teacher_disagreement_score":0.0036529468,"about_ca_system_score_codex":0.0007192646,"about_ca_system_score_gemma":0.0014081595,"threshold_uncertainty_score":0.019318879},"labels":[],"label_agreement":null},{"id":"W2081102656","doi":"10.1007/s00186-013-0432-y","title":"Accelerated modified policy iteration algorithms for Markov decision processes","year":2013,"lang":"en","type":"article","venue":"Mathematical Methods of Operations Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Markov decision process; Convergence (economics); Markov chain; Operator (biology); Mathematical optimization; Computer science; Markov process; Algorithm; Markov model; Partially observable Markov decision process; Mathematics; Machine learning; Statistics","score_opus":0.2714919902690478,"score_gpt":0.5351392337457531,"score_spread":0.2636472434767053,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2081102656","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011026879,0.00028349264,0.986426,0.0001960745,0.00009327294,0.00004528927,0.000029828454,0.00016096869,0.0017382308],"genre_scores_gemma":[0.5207879,0.00060653896,0.46699783,0.00024889436,0.00022656492,0.0006921596,0.00021691382,0.0002802026,0.009943084],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99855405,0.00076137827,0.00006892016,0.00016170673,0.00032306495,0.00013091227],"domain_scores_gemma":[0.9910773,0.00704517,0.00043116455,0.0003849308,0.0008085881,0.00025285364],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030598408,0.0011997621,0.0019644315,0.0009530113,0.00061725976,0.0013569206,0.002182053,0.0019080706,0.0048186714],"category_scores_gemma":[0.014144419,0.0011248736,0.0008908933,0.0008012196,0.0017264065,0.0020049985,0.0022298056,0.0030682744,0.0007200667],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010818467,0.000059854116,0.0002623534,0.0000662176,0.000039502906,0.00003166221,0.000054988588,0.9143505,0.00040391652,0.058126714,0.000925759,0.025570335],"study_design_scores_gemma":[0.000013913778,0.000010922674,0.000020793821,0.0000044063167,0.0000026397604,0.0000036059919,0.0000017325747,0.98526627,0.00007529038,0.014414419,0.000182551,0.0000034760003],"about_ca_topic_score_codex":0.005611167,"about_ca_topic_score_gemma":0.004629883,"teacher_disagreement_score":0.005611167,"about_ca_system_score_codex":0.0017556336,"about_ca_system_score_gemma":0.0030310696,"threshold_uncertainty_score":0.016182125},"labels":[],"label_agreement":null},{"id":"W2081239738","doi":"10.1109/adprl.2007.368191","title":"Opposition-Based Reinforcement Learning in the Management of Water Resources","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Exploit; Bellman equation; Opposition (politics); Artificial intelligence; Action learning; Operations research; Mathematical optimization; Engineering; Computer security; Mathematics; Law; Political science; Cooperative learning","score_opus":0.014481531006971042,"score_gpt":0.24468272489647908,"score_spread":0.23020119388950802,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2081239738","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022433331,0.00048585207,0.97311395,0.00034409075,0.000053644846,0.00003382667,0.000009112846,0.00008936353,0.003436844],"genre_scores_gemma":[0.9128029,0.00038878428,0.08356976,0.00014446197,0.000058532183,0.00011224175,0.000015228735,0.000025951116,0.00288223],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99938,0.0003470941,0.000021704649,0.000059334467,0.00014250386,0.000049440438],"domain_scores_gemma":[0.9990119,0.00068568997,0.000115554096,0.000041755953,0.0000992009,0.00004592472],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014340988,0.00033555576,0.000700085,0.000298289,0.00030087182,0.00057626533,0.0007886599,0.0008168786,0.0008704077],"category_scores_gemma":[0.0027608266,0.00024079044,0.00030912543,0.00034015116,0.0015550354,0.0009871616,0.0009508382,0.0009504044,0.00014507028],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009740288,0.000049114566,0.0004215297,0.00005616147,0.0000319445,0.00008911277,0.00007019615,0.91098416,0.0019821052,0.0541458,0.00045182713,0.0316207],"study_design_scores_gemma":[0.000017017583,0.000050033348,0.000053114032,0.000004957824,0.000004272827,0.000015461517,0.0000057938296,0.9809003,0.00038680588,0.017909413,0.0006469133,0.0000060150996],"about_ca_topic_score_codex":0.0015190771,"about_ca_topic_score_gemma":0.0011516014,"teacher_disagreement_score":0.0015190771,"about_ca_system_score_codex":0.00070479914,"about_ca_system_score_gemma":0.0005561204,"threshold_uncertainty_score":0.0075843334},"labels":[],"label_agreement":null},{"id":"W2082315153","doi":"10.1109/acii.2013.43","title":"Event-Driven Fuzzy Automata for Tracking Changes in the Emotional Behavior of Affective Agents","year":2013,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Automaton; Transition (genetics); Fuzzy logic; Arousal; Event (particle physics); Computer science; Pleasure; Dominance (genetics); Finite-state machine; Transition system; Artificial intelligence; Psychology; Cognitive psychology; Theoretical computer science; Social psychology; Algorithm; Physics","score_opus":0.04215215203548606,"score_gpt":0.3034167299870698,"score_spread":0.26126457795158375,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2082315153","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0411849,0.00016471077,0.9559541,0.00009818929,0.000043807515,0.000042336134,0.00008183712,0.000540329,0.0018898299],"genre_scores_gemma":[0.8869952,0.00011565078,0.11102239,0.000047465925,0.000012793983,0.00012932658,0.00009820387,0.000025738396,0.0015533176],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997323,0.00007724299,0.000022379252,0.00006957895,0.000074018884,0.000024408582],"domain_scores_gemma":[0.99934167,0.00038728505,0.00008908235,0.000056294877,0.00009870736,0.000026836185],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00050353736,0.00049323746,0.00041390164,0.00039047765,0.0002982688,0.00056646037,0.00078323757,0.0005514002,0.001146527],"category_scores_gemma":[0.0019673242,0.00022250607,0.00046966187,0.0002135835,0.0005508444,0.0006405846,0.0004752981,0.0007192495,0.00017643196],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010100912,0.00004547682,0.0019660355,0.000063798245,0.000049772007,0.00013285757,0.00017321945,0.9296031,0.0072187,0.023150839,0.00041032786,0.037084993],"study_design_scores_gemma":[0.0000040715163,0.000017597127,0.00014267307,0.000004480338,0.000005422088,0.000010530678,0.000006051038,0.9942233,0.0008734174,0.0044282107,0.0002789216,0.0000051976494],"about_ca_topic_score_codex":0.0054037087,"about_ca_topic_score_gemma":0.005682759,"teacher_disagreement_score":0.0054037087,"about_ca_system_score_codex":0.00075231364,"about_ca_system_score_gemma":0.00047470833,"threshold_uncertainty_score":0.010744512},"labels":[],"label_agreement":null},{"id":"W2085404981","doi":"10.3166/ria.20.311-344","title":"Étude de différentes combinaisons de comportements adaptatives","year":2006,"lang":"fr","type":"article","venue":"Revue d intelligence artificielle","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Psychology","score_opus":0.05580141030256195,"score_gpt":0.2918191637190429,"score_spread":0.23601775341648093,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2085404981","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7127192,0.0010477261,0.27202943,0.00027043163,0.00018931634,0.00015133916,0.00019274803,0.0013469637,0.012052914],"genre_scores_gemma":[0.95385,0.00019719775,0.042044226,0.00003731619,0.000025995309,0.000112047856,0.00012463513,0.00016728169,0.0034412616],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976052,0.00080185465,0.00017222312,0.00063139334,0.00059173343,0.00019757856],"domain_scores_gemma":[0.98138726,0.014493777,0.0004780753,0.0016151222,0.0016730056,0.00035278068],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034798027,0.0008744756,0.0013153953,0.0013752354,0.0008878777,0.0026210134,0.0013442776,0.0014272431,0.009033665],"category_scores_gemma":[0.014313203,0.0005659798,0.0014958733,0.0016448293,0.0012694316,0.0020151627,0.0009283808,0.0012368396,0.0008652342],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003755442,0.0011161176,0.015364583,0.0007361411,0.00096204015,0.00077853614,0.00094125414,0.5143411,0.07638272,0.024038073,0.0017590931,0.35982484],"study_design_scores_gemma":[0.00020104951,0.00069008855,0.010959473,0.000045454417,0.00034610563,0.00044972677,0.0003845756,0.93396914,0.041315656,0.008355298,0.0032073513,0.00007617862],"about_ca_topic_score_codex":0.003113722,"about_ca_topic_score_gemma":0.0024595475,"teacher_disagreement_score":0.009033665,"about_ca_system_score_codex":0.0012968174,"about_ca_system_score_gemma":0.00046582852,"threshold_uncertainty_score":0.030220568},"labels":[],"label_agreement":null},{"id":"W2086608109","doi":"10.1109/tnn.2011.2168422","title":"Hierarchical Approximate Policy Iteration With Binary-Tree State Space Decomposition","year":2011,"lang":"en","type":"article","venue":"IEEE Transactions on Neural Networks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Computer science; Markov decision process; Reinforcement learning; Kernel (algebra); State space; Mathematical optimization; Tree (set theory); Algorithm; Markov process; Artificial intelligence; Mathematics","score_opus":0.018154263890840724,"score_gpt":0.24563729892052735,"score_spread":0.22748303502968664,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2086608109","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006782828,0.00011697951,0.99202675,0.00004825805,0.000018210305,0.000029924184,0.00001928075,0.0002449503,0.00071274134],"genre_scores_gemma":[0.5534014,0.0002322468,0.44307312,0.00018011832,0.000039243074,0.00043565469,0.00024099094,0.00013403971,0.0022631977],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989479,0.00031576012,0.0000670562,0.00018434504,0.00036053202,0.00012438011],"domain_scores_gemma":[0.99855095,0.0007785615,0.00012844765,0.00013558223,0.0003238175,0.0000827261],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013841775,0.0008674485,0.0016087267,0.00061567366,0.00045196264,0.0009069126,0.0011728096,0.0011351494,0.0019524642],"category_scores_gemma":[0.0041424474,0.0006122989,0.0008120968,0.0006928028,0.0008196833,0.0012734734,0.0013004056,0.0016706637,0.00043762548],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009134349,0.00004820773,0.00044119696,0.000071886876,0.000028366252,0.000043051798,0.00007172691,0.9356131,0.0014789971,0.013958419,0.00068520824,0.0474685],"study_design_scores_gemma":[0.000004664442,0.000008265047,0.000013968399,0.0000014747977,0.0000012937268,0.0000029013265,0.0000015533458,0.9985885,0.00012360117,0.0011633369,0.000088955814,0.0000014657776],"about_ca_topic_score_codex":0.007478093,"about_ca_topic_score_gemma":0.0043564257,"teacher_disagreement_score":0.007478093,"about_ca_system_score_codex":0.00093150407,"about_ca_system_score_gemma":0.0020410314,"threshold_uncertainty_score":0.014869094},"labels":[],"label_agreement":null},{"id":"W2086813915","doi":"10.1007/s10846-009-9380-4","title":"A Reinforcement Learning Adaptive Fuzzy Controller for Differential Games","year":2009,"lang":"en","type":"article","venue":"Journal of Intelligent & Robotic Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Royal Military College of Canada","funders":"","keywords":"Reinforcement learning; Differential game; Computer science; Fuzzy logic; Differential (mechanical device); Reinforcement; Controller (irrigation); Control theory (sociology); Function (biology); Artificial intelligence; Control (management); Mathematical optimization; Engineering; Mathematics","score_opus":0.027291887146544194,"score_gpt":0.2659808781805624,"score_spread":0.2386889910340182,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2086813915","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029577952,0.00029649274,0.95923555,0.00027659308,0.00025023232,0.00013856939,0.000036043508,0.00046985998,0.00971873],"genre_scores_gemma":[0.9181769,0.00016804569,0.07583989,0.00017767352,0.000057310655,0.00021307568,0.00003986239,0.000028535811,0.0052987593],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999757,0.000043444914,0.000013671304,0.000054305157,0.00009231014,0.000039217783],"domain_scores_gemma":[0.99960023,0.00016371699,0.00003362724,0.000022432509,0.00014303542,0.000037045036],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000674751,0.0006140747,0.0010222319,0.00040078603,0.0005472015,0.00077633595,0.0014733395,0.0011161679,0.003077872],"category_scores_gemma":[0.001302429,0.0002821384,0.00044120706,0.00025205043,0.0006781944,0.00037976488,0.0010584525,0.00095069775,0.0003803868],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022544923,0.000200166,0.00048270033,0.00015337339,0.00008630068,0.00031011054,0.00013330509,0.8542918,0.011413537,0.023486005,0.002522012,0.10669533],"study_design_scores_gemma":[0.000044064276,0.00006406388,0.0000666504,0.0000051235925,0.000010056585,0.000020114809,0.000003474167,0.9972404,0.00045057162,0.001585337,0.0005034878,0.000006572126],"about_ca_topic_score_codex":0.0075752228,"about_ca_topic_score_gemma":0.0054799267,"teacher_disagreement_score":0.0075752228,"about_ca_system_score_codex":0.0007249978,"about_ca_system_score_gemma":0.00087295787,"threshold_uncertainty_score":0.015062213},"labels":[],"label_agreement":null},{"id":"W2094387729","doi":"10.1016/j.automatica.2009.07.008","title":"Natural actor–critic algorithms","year":2009,"lang":"en","type":"article","venue":"Automatica","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":569,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Temporal difference learning; Reinforcement learning; Computer science; Function approximation; Convergence (economics); Bellman equation; Mathematical proof; Function (biology); Stochastic gradient descent; Variance (accounting); Mathematics; Mathematical optimization; Artificial intelligence; Applied mathematics; Algorithm; Artificial neural network","score_opus":0.009214446772915929,"score_gpt":0.2557505181836024,"score_spread":0.24653607141068645,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2094387729","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004533079,0.0004114238,0.98612994,0.0002477291,0.00012222941,0.000034443063,0.00002717905,0.0005410488,0.0079528615],"genre_scores_gemma":[0.43471128,0.0006529115,0.5399045,0.000430031,0.00018860263,0.0003064238,0.00018646615,0.00028601987,0.02333382],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994937,0.00018848712,0.000025879966,0.00015673411,0.000101092424,0.000034174038],"domain_scores_gemma":[0.99873406,0.00077697396,0.000075842254,0.0001853443,0.00018275902,0.00004504412],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013363397,0.0008623444,0.000780511,0.0004885596,0.00057269126,0.00089771603,0.0012789469,0.0016165391,0.0069444003],"category_scores_gemma":[0.0038922394,0.0005689578,0.00050473295,0.0003782699,0.0012135209,0.0012508513,0.0010124308,0.0015519542,0.0016027221],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011022417,0.00010961889,0.0005981494,0.00020552584,0.00009454824,0.00009158626,0.00009331269,0.62110186,0.0034029773,0.165495,0.009209668,0.19948755],"study_design_scores_gemma":[0.000014674232,0.00001632886,0.000051143863,0.000008618609,0.000008045756,0.000024941508,0.0000049695977,0.9583105,0.0004703049,0.038647376,0.00243715,0.000005869494],"about_ca_topic_score_codex":0.0012351996,"about_ca_topic_score_gemma":0.0021751772,"teacher_disagreement_score":0.0069444003,"about_ca_system_score_codex":0.0005894248,"about_ca_system_score_gemma":0.0008228104,"threshold_uncertainty_score":0.023231268},"labels":[],"label_agreement":null},{"id":"W2095564494","doi":"10.1142/s0219525911002998","title":"AN EMPIRICAL STUDY OF POTENTIAL-BASED REWARD SHAPING AND ADVICE IN COMPLEX, MULTI-AGENT SYSTEMS","year":2011,"lang":"en","type":"article","venue":"Advances in Complex Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":79,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Context (archaeology); Reward system; Nash equilibrium; Domain (mathematical analysis); Artificial intelligence; Machine learning; Psychology; Mathematical optimization; Mathematics","score_opus":0.1324506532271785,"score_gpt":0.3578127690604147,"score_spread":0.2253621158332362,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2095564494","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95062655,0.0007417652,0.043058034,0.0010479077,0.000015523343,0.00006967784,0.00009524567,0.0000837109,0.004261613],"genre_scores_gemma":[0.9964637,0.000082602004,0.0031810214,0.000023659259,0.0000045761276,0.000014046539,0.000028177612,0.000006946979,0.00019525667],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981856,0.0011779531,0.00008606816,0.00018926628,0.00026688326,0.00009421562],"domain_scores_gemma":[0.9064002,0.08117667,0.0053832154,0.004009358,0.0019275103,0.0011030183],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005362243,0.0003629373,0.0004174017,0.0005797269,0.0005072165,0.0009380387,0.0009274998,0.0011657798,0.0020686127],"category_scores_gemma":[0.087692544,0.00027746003,0.00025828232,0.0006335902,0.0018291449,0.0025099348,0.00095128815,0.0015219504,0.00014401288],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051910913,0.0007315437,0.091400474,0.00047385707,0.00024038654,0.00028602732,0.0011655443,0.78393435,0.0026666753,0.054207314,0.0016230836,0.06275163],"study_design_scores_gemma":[0.00009211767,0.00039348335,0.030933263,0.000046560865,0.00003144073,0.00014549211,0.00039518278,0.93173325,0.0010469165,0.03371331,0.001430551,0.000038430764],"about_ca_topic_score_codex":0.0030378236,"about_ca_topic_score_gemma":0.001890323,"teacher_disagreement_score":0.005362243,"about_ca_system_score_codex":0.00081827666,"about_ca_system_score_gemma":0.0004905882,"threshold_uncertainty_score":0.028358579},"labels":[],"label_agreement":null},{"id":"W2096976789","doi":"","title":"Theoretical Analysis of Heuristic Search Methods for Online POMDPs.","year":2008,"lang":"en","type":"article","venue":"PubMed","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval; McGill University; McGill University Health Centre","funders":"","keywords":"Heuristics; Scalability; Computer science; Heuristic; Online search; Mathematical optimization; Online algorithm; Artificial intelligence; Algorithm; Mathematics; Information retrieval","score_opus":0.07173965773491928,"score_gpt":0.35214968483620557,"score_spread":0.28041002710128626,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2096976789","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0034021214,0.00071383885,0.98899055,0.00044730402,0.00004017571,0.00007513588,0.000078190526,0.000160885,0.00609177],"genre_scores_gemma":[0.45606646,0.002784038,0.53158104,0.0006563851,0.00031060216,0.0013640734,0.00056292984,0.0004230313,0.0062514883],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9947684,0.0024427834,0.00022501386,0.00058236776,0.001587253,0.00039420876],"domain_scores_gemma":[0.9653887,0.030341372,0.001490114,0.0012742564,0.0010663597,0.00043921926],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0073911627,0.002104433,0.001618478,0.0021274316,0.0011848283,0.003080457,0.003011165,0.0023212899,0.006784416],"category_scores_gemma":[0.0361933,0.0010536229,0.0021262062,0.0018618354,0.0049530054,0.0048947427,0.0030542312,0.00465709,0.0008588535],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008131431,0.000076341705,0.0005646389,0.0004190637,0.00008676409,0.00006428549,0.00017009715,0.5130867,0.00042647545,0.4568828,0.0016775558,0.026463998],"study_design_scores_gemma":[0.000025086858,0.000048277892,0.00007947581,0.00007932194,0.000021486247,0.00002741876,0.000030182591,0.7313156,0.0003307463,0.2664692,0.0015616379,0.0000115004395],"about_ca_topic_score_codex":0.0033662193,"about_ca_topic_score_gemma":0.0027647666,"teacher_disagreement_score":0.0073911627,"about_ca_system_score_codex":0.003558826,"about_ca_system_score_gemma":0.0034222552,"threshold_uncertainty_score":0.039088726},"labels":[],"label_agreement":null},{"id":"W2097113539","doi":"10.1613/jair.898","title":"Accelerating Reinforcement Learning through Implicit Imitation","year":2003,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":176,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; University of British Columbia","funders":"Instituto de Ciencias del Mar y Limnología, Universidad Nacional Autónoma de México; Natural Sciences and Engineering Research Council of Canada","keywords":"Imitation; Reinforcement learning; Observability; Computer science; Convergence (economics); Action (physics); Space (punctuation); Artificial intelligence; State space; Reinforcement; Value (mathematics); Human–computer interaction; Machine learning; Psychology; Mathematics; Social psychology","score_opus":0.25418857368125086,"score_gpt":0.4324161648659641,"score_spread":0.17822759118471326,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2097113539","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11907723,0.00008316888,0.87716705,0.00017519505,0.00002136296,0.00005266925,0.000016486618,0.00073772244,0.0026690932],"genre_scores_gemma":[0.9457831,0.00004690671,0.052645832,0.000033660657,0.0000081672315,0.00006776935,0.000019388435,0.00002341401,0.0013719057],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99938524,0.00023551816,0.000028106324,0.00010860332,0.00015554958,0.00008695967],"domain_scores_gemma":[0.9958956,0.002361286,0.00047243264,0.00084554614,0.00027995728,0.00014525086],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014046049,0.00065310026,0.00065463484,0.00024137936,0.0002388503,0.00050026865,0.0013743608,0.0007705024,0.0013653109],"category_scores_gemma":[0.006951879,0.00033900738,0.00030973874,0.00021462707,0.0012728894,0.0013757927,0.001438488,0.0012365663,0.00022306474],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021967253,0.00015043486,0.0016646644,0.00008514465,0.00003645266,0.00013549013,0.00018259142,0.9001032,0.009209661,0.03314264,0.00039072844,0.05467931],"study_design_scores_gemma":[0.000024732211,0.00006284487,0.00009283708,0.0000030322615,0.0000040610134,0.000016029959,0.000004221215,0.99240136,0.0010912344,0.006090937,0.00020483858,0.0000039156275],"about_ca_topic_score_codex":0.0016659484,"about_ca_topic_score_gemma":0.0013328297,"teacher_disagreement_score":0.0016659484,"about_ca_system_score_codex":0.0005211018,"about_ca_system_score_gemma":0.00078687817,"threshold_uncertainty_score":0.0074282885},"labels":[],"label_agreement":null},{"id":"W2097575529","doi":"","title":"Reinforcement Learning using Kernel-Based Stochastic Factorization","year":2011,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Kernel (algebra); Factorization; Markov decision process; Scalability; Matrix decomposition; Artificial intelligence; Machine learning; Algorithm; Markov process; Mathematics; Eigenvalues and eigenvectors","score_opus":0.0600288240839159,"score_gpt":0.2556550448293221,"score_spread":0.1956262207454062,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2097575529","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007439903,0.00007324456,0.9918434,0.00006514456,0.000011308849,0.000023911036,0.000010891972,0.00017298284,0.00035911734],"genre_scores_gemma":[0.7889496,0.00018868245,0.20889692,0.00010158531,0.000040159022,0.00020258699,0.00010345938,0.00006846606,0.0014485035],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991442,0.0003605451,0.000052056945,0.00013739897,0.00021289535,0.000092978175],"domain_scores_gemma":[0.9958442,0.0029724457,0.0003393929,0.00021512712,0.00049930904,0.00012952134],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021852062,0.00089933455,0.0015067863,0.00051881507,0.00035658714,0.0007065295,0.0009833327,0.0010080036,0.0014488837],"category_scores_gemma":[0.00814984,0.00051171915,0.00060292706,0.00043120937,0.001255769,0.0013318767,0.0010720867,0.001677597,0.0003012504],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005827739,0.00003718767,0.00039118255,0.000049682134,0.000028252172,0.000035009994,0.00003841793,0.9628912,0.001020142,0.0118178185,0.00036613643,0.02326654],"study_design_scores_gemma":[0.0000055137925,0.00001066909,0.000018237968,0.0000017413407,0.0000015200545,0.0000034298725,0.0000011521377,0.99749833,0.00012636186,0.0022781794,0.000052979252,0.0000019265658],"about_ca_topic_score_codex":0.0049099606,"about_ca_topic_score_gemma":0.0031492813,"teacher_disagreement_score":0.0049099606,"about_ca_system_score_codex":0.0011238711,"about_ca_system_score_gemma":0.0015565145,"threshold_uncertainty_score":0.011556625},"labels":[],"label_agreement":null},{"id":"W2098152875","doi":"10.1145/1390156.1390199","title":"Reinforcement learning in the presence of rare events","year":2008,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Reinforcement learning; Computer science; Reinforcement; Artificial intelligence; Psychology; Social psychology","score_opus":0.029295139754644966,"score_gpt":0.25547336988383174,"score_spread":0.22617823012918678,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2098152875","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17028263,0.00039023836,0.82543975,0.00084538467,0.00006240057,0.000054415435,0.000045301673,0.00042341437,0.0024565079],"genre_scores_gemma":[0.9593086,0.00011781614,0.03944663,0.000113239155,0.00004079888,0.00006016507,0.000031484327,0.00003313723,0.0008481294],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979353,0.001166419,0.00007876464,0.0003119685,0.0003208163,0.00018679441],"domain_scores_gemma":[0.9752204,0.020605665,0.001762439,0.00090287917,0.0010111796,0.0004973779],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054237964,0.0006053114,0.0013352133,0.0005195242,0.000511083,0.0012071765,0.0013546603,0.0010023783,0.0007559364],"category_scores_gemma":[0.028769242,0.00056154083,0.00033707684,0.00039222252,0.0021518255,0.001950125,0.0012327073,0.0017954482,0.00012317827],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017216441,0.00004525064,0.002029529,0.000048504364,0.000050349478,0.00011921444,0.00008374515,0.9518221,0.00063755875,0.029042138,0.0003873829,0.015562025],"study_design_scores_gemma":[0.000022966999,0.000019768338,0.00017690397,0.0000043956275,0.0000054950006,0.000012908481,0.000006364387,0.9818651,0.00021791228,0.01754811,0.000115620925,0.0000044664666],"about_ca_topic_score_codex":0.0039051569,"about_ca_topic_score_gemma":0.0028098284,"teacher_disagreement_score":0.0054237964,"about_ca_system_score_codex":0.0014645967,"about_ca_system_score_gemma":0.0012105607,"threshold_uncertainty_score":0.02868414},"labels":[],"label_agreement":null},{"id":"W2099089474","doi":"","title":"PAC-Bayesian Model Selection for Reinforcement Learning","year":2010,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Leverage (statistics); Computer science; Bayesian probability; Correctness; Selection (genetic algorithm); Artificial intelligence; Machine learning; Bayesian inference; Model selection; Algorithm","score_opus":0.014450719471556406,"score_gpt":0.25735404762560055,"score_spread":0.24290332815404414,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2099089474","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010390287,0.0004909728,0.99480164,0.00028749046,0.000040277413,0.000027521453,0.000040017138,0.0002384164,0.003034682],"genre_scores_gemma":[0.49048558,0.0022011315,0.49621418,0.0012325612,0.0006668601,0.0010266852,0.0006086611,0.00078111124,0.006783193],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9914743,0.003786962,0.0003079082,0.0010478862,0.00291565,0.00046725487],"domain_scores_gemma":[0.963535,0.030183773,0.0012803234,0.0019985735,0.0023608452,0.00064135296],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008540191,0.0025409625,0.0026386608,0.0015713023,0.0010080473,0.0031012269,0.0033158152,0.0022469724,0.0060171816],"category_scores_gemma":[0.04576465,0.0014842426,0.001299115,0.0017203924,0.0036982691,0.006067636,0.003609906,0.008434514,0.0015657217],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015445768,0.00009131429,0.00050124695,0.00026394936,0.00011796443,0.00010126935,0.00016293241,0.5397377,0.0008634031,0.39205536,0.0055555236,0.06039473],"study_design_scores_gemma":[0.000012107061,0.000020666126,0.000049169535,0.000025193367,0.000008643784,0.000020079427,0.000005336562,0.8285568,0.0003339677,0.16979136,0.0011655162,0.000011273597],"about_ca_topic_score_codex":0.00495114,"about_ca_topic_score_gemma":0.0043247,"teacher_disagreement_score":0.008540191,"about_ca_system_score_codex":0.004316776,"about_ca_system_score_gemma":0.0035796342,"threshold_uncertainty_score":0.04516542},"labels":[],"label_agreement":null},{"id":"W2099765827","doi":"10.1109/icsmc.2007.4414257","title":"Active exploratory q-learning for large problems","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Curse of dimensionality; Computer science; Artificial intelligence; Bellman equation; Machine learning; Robot; Action (physics); Mathematical optimization; Mathematics","score_opus":0.026113218419187033,"score_gpt":0.2715003364265079,"score_spread":0.24538711800732085,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2099765827","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009295532,0.0005224517,0.987711,0.00027151447,0.000020448078,0.00005409133,0.00001056216,0.00019127507,0.0019230979],"genre_scores_gemma":[0.6572089,0.0008200164,0.33688086,0.00033529382,0.00012017918,0.0007686712,0.00009451729,0.00012805298,0.003643556],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998896,0.00064306794,0.00004682159,0.00011114598,0.00021958059,0.00008339229],"domain_scores_gemma":[0.99050015,0.008069236,0.0003425439,0.00036883418,0.00049023464,0.00022901142],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040514297,0.0008487946,0.0014453607,0.000670585,0.0005626418,0.0008758907,0.0016305596,0.001262443,0.0025829352],"category_scores_gemma":[0.011106034,0.00049972197,0.0005680889,0.0006117043,0.0016742178,0.001385724,0.0019061483,0.0016783343,0.00034285436],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014113139,0.00012812372,0.0007121281,0.00021225908,0.000059215537,0.00011291523,0.00015145636,0.8395384,0.0013042075,0.073426075,0.001599284,0.08261476],"study_design_scores_gemma":[0.000030349775,0.00004212082,0.000046432207,0.000008640112,0.0000043417754,0.000013086493,0.0000074548307,0.97463155,0.0001988534,0.024461623,0.0005516451,0.000003900005],"about_ca_topic_score_codex":0.0015656971,"about_ca_topic_score_gemma":0.0013262383,"teacher_disagreement_score":0.0040514297,"about_ca_system_score_codex":0.00081713736,"about_ca_system_score_gemma":0.0010961159,"threshold_uncertainty_score":0.02142626},"labels":[],"label_agreement":null},{"id":"W2100785108","doi":"10.7551/mitpress/7503.003.0062","title":"Bayesian Policy Gradient Algorithms","year":2007,"lang":"en","type":"book-chapter","venue":"The MIT Press eBooks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":70,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Bayesian probability; Computer science; Algorithm; Artificial intelligence","score_opus":0.053027897781275965,"score_gpt":0.27853266809186156,"score_spread":0.22550477031058558,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2100785108","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00076060306,0.0014738813,0.98704463,0.00029971742,0.0001227049,0.00010271119,0.00012571123,0.0008005696,0.009269485],"genre_scores_gemma":[0.11187105,0.0045561036,0.84924394,0.0006432585,0.00036425702,0.0009476403,0.0008407934,0.00080310414,0.030729918],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99826616,0.00062904716,0.00008855371,0.00028915273,0.00056727714,0.00015980857],"domain_scores_gemma":[0.9980646,0.0011265056,0.00015672867,0.00017882,0.000395037,0.000078390236],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025595997,0.0021139171,0.002276008,0.0013743764,0.00086983724,0.0026615153,0.0029694652,0.0024580094,0.018332796],"category_scores_gemma":[0.009630341,0.0011917913,0.000903022,0.0017315906,0.0013544686,0.0031835262,0.0023930022,0.0029240518,0.006969027],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011860111,0.00014295196,0.0006347791,0.00040059607,0.00012033847,0.000056973964,0.00009748002,0.33000398,0.0006483807,0.25375494,0.024764217,0.38925678],"study_design_scores_gemma":[0.00006127701,0.000031449767,0.00016720624,0.00010523383,0.000026893124,0.000046720168,0.000018345607,0.7847727,0.0007908118,0.18494238,0.029008916,0.00002805399],"about_ca_topic_score_codex":0.005256637,"about_ca_topic_score_gemma":0.005057265,"teacher_disagreement_score":0.018332796,"about_ca_system_score_codex":0.0019443404,"about_ca_system_score_gemma":0.002691027,"threshold_uncertainty_score":0.061329365},"labels":[],"label_agreement":null},{"id":"W2101911972","doi":"10.1109/robot.2010.5509717","title":"Apprenticeship learning via soft local homomorphisms","year":2010,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Markov decision process; Computer science; Reinforcement learning; State space; Artificial intelligence; Robot; Process (computing); Markov process; Space (punctuation); Homomorphism; Apprenticeship; Machine learning; Mathematics; Discrete mathematics; Statistics","score_opus":0.008005428190261192,"score_gpt":0.21867840387644316,"score_spread":0.21067297568618196,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2101911972","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17289092,0.00015252593,0.8226511,0.00034725422,0.000029309927,0.000111772904,0.0000310328,0.00075956516,0.0030265802],"genre_scores_gemma":[0.96876425,0.000040112034,0.029797891,0.00007245432,0.000012333943,0.000101272046,0.000033592554,0.000023245859,0.0011548146],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990839,0.00042950676,0.00004077801,0.0002188416,0.0001303543,0.00009679075],"domain_scores_gemma":[0.9957885,0.0026835797,0.00029995907,0.0006077836,0.00032865198,0.00029144486],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015566687,0.00050968555,0.0010465686,0.00043918265,0.00039171826,0.00066375587,0.00155298,0.0009607542,0.0025763223],"category_scores_gemma":[0.008541621,0.00034901508,0.00059663266,0.0003467103,0.0017215918,0.0021035841,0.0022802555,0.0015603482,0.00040646942],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030214898,0.0005446013,0.0034122884,0.00013159036,0.00007430567,0.0002098747,0.00038071728,0.7590613,0.004213738,0.027231839,0.0010396689,0.20339787],"study_design_scores_gemma":[0.000014891905,0.000079587226,0.00015537965,0.000005950008,0.0000039930237,0.000022362094,0.000022211569,0.98714954,0.00091687945,0.011473351,0.0001508219,0.0000049920086],"about_ca_topic_score_codex":0.0014302541,"about_ca_topic_score_gemma":0.0009181009,"teacher_disagreement_score":0.0025763223,"about_ca_system_score_codex":0.0006624341,"about_ca_system_score_gemma":0.00066477986,"threshold_uncertainty_score":0.008618653},"labels":[],"label_agreement":null},{"id":"W2101984404","doi":"","title":"VDCBPI: an Approximate Scalable Algorithm for Large POMDPs","year":2004,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":65,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Scalability; Markov decision process; Curse of dimensionality; Computer science; Bounded function; Markov process; Mathematical optimization; Observable; Algorithm; Theoretical computer science; Mathematics; Artificial intelligence","score_opus":0.018618968701239788,"score_gpt":0.26621215775492185,"score_spread":0.24759318905368205,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2101984404","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0042283875,0.00016505738,0.9900107,0.0001804993,0.00006939404,0.00013092376,0.000108964734,0.0030926915,0.0020134577],"genre_scores_gemma":[0.16031626,0.00017644356,0.83597684,0.00022352445,0.00004580975,0.0005815393,0.00046910948,0.00039728967,0.0018131227],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993954,0.00011116359,0.00003553686,0.00014033055,0.00023247582,0.00008516131],"domain_scores_gemma":[0.9985776,0.0007818908,0.00008821865,0.00026736647,0.00018964947,0.00009532982],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012441142,0.0011027919,0.0012985404,0.0008297948,0.0007584221,0.0011273099,0.0026247057,0.0013844623,0.0057835053],"category_scores_gemma":[0.0046022264,0.0007252821,0.00069384545,0.0008792078,0.0008230268,0.0016647617,0.0027255204,0.0023386735,0.0010933774],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020650185,0.00012493637,0.0008232992,0.00020796522,0.00006696322,0.000103363935,0.00007376865,0.5949464,0.0024801337,0.027211916,0.009299424,0.3644554],"study_design_scores_gemma":[0.000040898973,0.0000145793065,0.000031660806,0.000006885818,0.0000040122586,0.000016454622,0.000006494576,0.99029595,0.00060715055,0.007825111,0.0011464888,0.000004294762],"about_ca_topic_score_codex":0.0074950075,"about_ca_topic_score_gemma":0.007877448,"teacher_disagreement_score":0.0074950075,"about_ca_system_score_codex":0.0013517863,"about_ca_system_score_gemma":0.004133764,"threshold_uncertainty_score":0.019347727},"labels":[],"label_agreement":null},{"id":"W2102847492","doi":"10.1145/1390156.1390286","title":"Apprenticeship learning using linear programming","year":2008,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":194,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Apprenticeship; Markov decision process; Computer science; Linear programming; Solver; Frame (networking); Markov process; Mathematical optimization; Process (computing); Markov chain; Artificial intelligence; Machine learning; Algorithm; Mathematics; Statistics","score_opus":0.06668564856124883,"score_gpt":0.2809145654987666,"score_spread":0.21422891693751778,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2102847492","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014893091,0.00019059029,0.9809168,0.00022817131,0.000021224736,0.0000673989,0.00001861235,0.0005753303,0.0030887718],"genre_scores_gemma":[0.545786,0.0003493213,0.44445112,0.00036857402,0.00006750543,0.00051863183,0.00018954552,0.00020289393,0.008066266],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989291,0.00043134327,0.00004962811,0.0002936368,0.00017888428,0.00011744212],"domain_scores_gemma":[0.9971083,0.002159883,0.00015469483,0.0001615151,0.00027878632,0.0001369018],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016989738,0.0012599779,0.0011634905,0.0005045889,0.0005508468,0.0011114156,0.0017544256,0.001565982,0.004719628],"category_scores_gemma":[0.006134431,0.0007459231,0.0007552595,0.00046894766,0.0012889894,0.0019088336,0.0024867968,0.0028746077,0.00085898367],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010416544,0.00026774345,0.0009396519,0.00012108123,0.000050764393,0.00012777121,0.00016991468,0.824541,0.0013002998,0.033219792,0.0019794046,0.13717848],"study_design_scores_gemma":[0.000012490598,0.00004377675,0.000038441005,0.000009801571,0.000004383404,0.000018589102,0.000013386011,0.98110896,0.00071622967,0.017379038,0.0006501589,0.000004834195],"about_ca_topic_score_codex":0.002640314,"about_ca_topic_score_gemma":0.0025908365,"teacher_disagreement_score":0.004719628,"about_ca_system_score_codex":0.0008756542,"about_ca_system_score_gemma":0.0014019074,"threshold_uncertainty_score":0.015788674},"labels":[],"label_agreement":null},{"id":"W2103568863","doi":"10.48550/arxiv.1407.0449","title":"Classification-based Approximate Policy Iteration: Experiments and Extended Discussions","year":2014,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Exploit; Estimator; Computer science; Reinforcement learning; Bellman equation; Classifier (UML); Nonparametric statistics; Function (biology); Temporal difference learning; Mathematical optimization; Machine learning; Artificial intelligence; Algorithm; Mathematics; Econometrics; Statistics","score_opus":0.08066957996727817,"score_gpt":0.23344313061038263,"score_spread":0.15277355064310447,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2103568863","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5574746,0.012243445,0.37767524,0.0053605842,0.0010816159,0.0013580328,0.0022192248,0.0045500197,0.03803715],"genre_scores_gemma":[0.8947501,0.001001046,0.0974792,0.0005470537,0.000111189795,0.0006627474,0.0015044388,0.00030117598,0.0036430706],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9962327,0.0017746026,0.00029332604,0.0005140254,0.0007849573,0.00040031542],"domain_scores_gemma":[0.9655124,0.025755698,0.001311165,0.004150382,0.002558936,0.00071136514],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0072131064,0.0013802522,0.0019302149,0.00067854696,0.0009366816,0.0013499678,0.0020086064,0.002631445,0.0060442262],"category_scores_gemma":[0.04318187,0.0004448661,0.0007785911,0.0013308142,0.0013377977,0.0029027548,0.0019077383,0.0034474176,0.0010902586],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002484642,0.003031871,0.006157955,0.0012741613,0.00026669182,0.00017538818,0.0003215718,0.80948895,0.0028355527,0.017706417,0.017640911,0.1386159],"study_design_scores_gemma":[0.00024655848,0.00039466383,0.0008140261,0.000060877417,0.000034723347,0.000043001277,0.00009256813,0.9799549,0.0025118059,0.013713542,0.002106828,0.000026555887],"about_ca_topic_score_codex":0.011546617,"about_ca_topic_score_gemma":0.0068260203,"teacher_disagreement_score":0.011546617,"about_ca_system_score_codex":0.0014680352,"about_ca_system_score_gemma":0.0020678164,"threshold_uncertainty_score":0.038146973},"labels":[],"label_agreement":null},{"id":"W2104641222","doi":"10.1177/105971230501300301","title":"Reinforcement Learning for RoboCup Soccer Keepaway","year":2005,"lang":"en","type":"article","venue":"Adaptive Behavior","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":390,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Generality; Computer science; Artificial intelligence; Learning classifier system; Benchmark (surveying); Machine learning; Task (project management); Psychology","score_opus":0.03632614507992138,"score_gpt":0.28627491977920116,"score_spread":0.24994877469927979,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2104641222","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2652726,0.00034492722,0.7282676,0.0006733661,0.00005695808,0.00010858617,0.00006104542,0.0012294626,0.003985442],"genre_scores_gemma":[0.97999257,0.000044899505,0.019073704,0.000047125955,0.000007936806,0.000059254355,0.000032659573,0.000015935399,0.00072590227],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997601,0.00010346859,0.000012195177,0.000042662152,0.000046487745,0.00003515376],"domain_scores_gemma":[0.9987208,0.0008347277,0.00012556305,0.00007727383,0.00014921749,0.00009237114],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008475916,0.00047181008,0.00057333865,0.00022018819,0.00027382685,0.00034695925,0.0006829743,0.0005091711,0.0014008211],"category_scores_gemma":[0.0033196595,0.00022341701,0.00020258634,0.00013567063,0.00068168837,0.0004538231,0.0005990446,0.00092816015,0.00013673535],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000062944644,0.000051210005,0.00067162025,0.000022561657,0.000014153801,0.000027643118,0.00003074422,0.97835857,0.00062702026,0.0021565172,0.00028962156,0.017687354],"study_design_scores_gemma":[0.000009836617,0.000017328079,0.000055241264,0.0000016366384,0.0000017042025,0.0000025686227,0.0000032757487,0.9983328,0.00023069937,0.0012571001,0.00008631531,0.0000015422432],"about_ca_topic_score_codex":0.007358919,"about_ca_topic_score_gemma":0.0050808974,"teacher_disagreement_score":0.007358919,"about_ca_system_score_codex":0.0008225125,"about_ca_system_score_gemma":0.0009632353,"threshold_uncertainty_score":0.014632165},"labels":[],"label_agreement":null},{"id":"W2105114545","doi":"","title":"Interval Estimation for Reinforcement-Learning Algorithms in Continuous-State Domains","year":2010,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Bootstrapping (finance); Markov decision process; Computer science; Machine learning; Artificial intelligence; Confidence interval; Value (mathematics); Algorithm; Markov process; Statistics; Mathematics; Econometrics","score_opus":0.012292104332855507,"score_gpt":0.2686448487087605,"score_spread":0.256352744375905,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2105114545","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0033433437,0.00022416552,0.99557614,0.000080233396,0.000018494871,0.00002353626,0.000021501844,0.00029840702,0.00041428406],"genre_scores_gemma":[0.45577544,0.0005582791,0.54165053,0.0001549353,0.00011225431,0.0003862028,0.00027474534,0.00029924963,0.0007884523],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9949126,0.0025045252,0.0003883963,0.00073084375,0.0012200185,0.00024351692],"domain_scores_gemma":[0.919885,0.06983439,0.0029068424,0.0030653079,0.0035046095,0.00080391794],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01062816,0.0013437823,0.0017454745,0.0018228551,0.00056052645,0.0024940928,0.0028538895,0.0017794829,0.0031879295],"category_scores_gemma":[0.10347772,0.0007709941,0.0009818746,0.0015013503,0.0023348117,0.0037212067,0.0025718918,0.0051181996,0.00070926914],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002577494,0.000095321,0.0018810299,0.00019991813,0.00007642811,0.00007482716,0.00023714753,0.8490941,0.00086261577,0.06904817,0.00096871454,0.077204004],"study_design_scores_gemma":[0.000013786947,0.000025910358,0.00008906359,0.000024158664,0.000005555406,0.000011256228,0.0000072344783,0.97540104,0.00035235664,0.023857998,0.00020271329,0.000008938629],"about_ca_topic_score_codex":0.00363801,"about_ca_topic_score_gemma":0.0014395171,"teacher_disagreement_score":0.01062816,"about_ca_system_score_codex":0.0019100787,"about_ca_system_score_gemma":0.0013423292,"threshold_uncertainty_score":0.056207776},"labels":[],"label_agreement":null},{"id":"W2105127884","doi":"10.1109/ia.2009.4927496","title":"How emotional mechanism helps episodic learning in a cognitive agent","year":2009,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke; Université du Québec à Montréal","funders":"Fonds Québécois de la Recherche sur la Nature et les Technologies","keywords":"Episodic memory; Mechanism (biology); Computer science; Cognition; Memory consolidation; TRACE (psycholinguistics); Engram; Encoding (memory); Cognitive architecture; Cognitive science; Cognitive psychology; Cognitive model; Artificial intelligence; Psychology; Neuroscience; Hippocampus","score_opus":0.02740953433025567,"score_gpt":0.25763938789719043,"score_spread":0.23022985356693476,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2105127884","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27664468,0.00042361606,0.69342834,0.0030535746,0.0002720482,0.00008141009,0.000046581394,0.0009482131,0.025101539],"genre_scores_gemma":[0.9133737,0.0001898887,0.08027109,0.00021368785,0.000046994963,0.000043081516,0.000030291894,0.000036064434,0.0057951678],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998282,0.00004787063,0.000013714694,0.00005172595,0.000033081855,0.00002527995],"domain_scores_gemma":[0.9994609,0.0002050275,0.00006380196,0.00010071785,0.00009477479,0.00007488699],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004359436,0.00025905843,0.00018713267,0.0001465514,0.0003815946,0.0010082615,0.00064550806,0.0007545095,0.0028828995],"category_scores_gemma":[0.002332196,0.0001507086,0.0002838636,0.00008206108,0.00081936934,0.002205348,0.00070739794,0.00060743757,0.00041638076],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039279708,0.00040357764,0.009031116,0.00041706144,0.00022927104,0.0011964764,0.0024176447,0.14268176,0.08644844,0.52961916,0.0041382154,0.22302449],"study_design_scores_gemma":[0.00015693608,0.00044224376,0.0029627087,0.000060936985,0.00026825882,0.00081961334,0.00044759782,0.66653395,0.046419468,0.2554809,0.026331574,0.00007578952],"about_ca_topic_score_codex":0.0005784006,"about_ca_topic_score_gemma":0.0004445295,"teacher_disagreement_score":0.0028828995,"about_ca_system_score_codex":0.0002408405,"about_ca_system_score_gemma":0.00031242007,"threshold_uncertainty_score":0.00964427},"labels":[],"label_agreement":null},{"id":"W2105148789","doi":"10.1145/545056.545107","title":"Being the new guy in an experienced team","year":2002,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Flexibility (engineering); Computer science; Process (computing); Action (physics); Focus (optics); Artificial intelligence; Scratch; Human–computer interaction; Knowledge management; Management","score_opus":0.02793981256670081,"score_gpt":0.26034342719760006,"score_spread":0.23240361463089926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2105148789","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2485539,0.0002312757,0.69413966,0.0026206062,0.000295655,0.00011486073,0.000030488778,0.000637269,0.053376302],"genre_scores_gemma":[0.8366147,0.00013666939,0.13753417,0.0005728493,0.00006503077,0.000069577596,0.000053580923,0.000097223696,0.02485613],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99911076,0.00032211814,0.000023692015,0.00023450781,0.00018384641,0.0001249789],"domain_scores_gemma":[0.99887997,0.00022235487,0.00016318313,0.0002875176,0.00012471776,0.00032236302],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008186997,0.00044136186,0.0003312524,0.00015940923,0.00070253725,0.0013477093,0.0011478963,0.0010629805,0.0038423284],"category_scores_gemma":[0.0027117906,0.00021794678,0.00038952776,0.0001154603,0.0013762972,0.0017699214,0.0022499377,0.0011034565,0.0010232052],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006414449,0.0010455503,0.01459286,0.00037578662,0.0003120179,0.0024155953,0.0077390107,0.10836193,0.046260003,0.26009312,0.015532473,0.54263014],"study_design_scores_gemma":[0.00016903479,0.001956716,0.006287122,0.00019560583,0.00028765196,0.0037982517,0.0033976922,0.5919127,0.032677624,0.23634623,0.12274801,0.00022331368],"about_ca_topic_score_codex":0.0005580537,"about_ca_topic_score_gemma":0.0009068906,"teacher_disagreement_score":0.0038423284,"about_ca_system_score_codex":0.00034386764,"about_ca_system_score_gemma":0.00042076508,"threshold_uncertainty_score":0.012853861},"labels":[],"label_agreement":null},{"id":"W2106919681","doi":"10.1109/noms.2010.5488472","title":"Towards adaptive policy-based management","year":2010,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science","score_opus":0.012975108101956432,"score_gpt":0.25474589984564394,"score_spread":0.2417707917436875,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2106919681","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006551902,0.00028839766,0.9873723,0.00086974766,0.000054214575,0.000040949813,0.000028678862,0.0005311202,0.004262609],"genre_scores_gemma":[0.61321247,0.0011494991,0.37958708,0.00058469235,0.00017028868,0.00035596322,0.00016794432,0.00014903625,0.004623006],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9988127,0.00042382078,0.00006884137,0.000225312,0.00032514962,0.00014427066],"domain_scores_gemma":[0.9984669,0.00073219393,0.00017510673,0.00031286187,0.0002045987,0.000108320535],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021409448,0.0008043529,0.0006928873,0.00048992346,0.00046714768,0.0018804183,0.0016207009,0.001434676,0.002185544],"category_scores_gemma":[0.0047497824,0.00054428354,0.0006206976,0.0005587395,0.0014867428,0.0021791623,0.002364928,0.0028604975,0.00056612195],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000058563557,0.00010482263,0.0006529892,0.00010013268,0.00006560484,0.00013318527,0.00025775068,0.7532731,0.0024270935,0.1846911,0.0023019058,0.055933785],"study_design_scores_gemma":[0.000011054944,0.000014079772,0.00004928046,0.000012154456,0.000006236384,0.000016135642,0.00002248534,0.9096262,0.0004840012,0.086454734,0.0032967573,0.00000682252],"about_ca_topic_score_codex":0.0029979884,"about_ca_topic_score_gemma":0.002183991,"teacher_disagreement_score":0.0029979884,"about_ca_system_score_codex":0.0011938319,"about_ca_system_score_gemma":0.0017429614,"threshold_uncertainty_score":0.011322498},"labels":[],"label_agreement":null},{"id":"W2107479312","doi":"10.1109/noms.2008.4575242","title":"Reinforcement learning in policy-driven autonomic management","year":2008,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Reinforcement learning; Autonomic computing; Computer science; Context (archaeology); Set (abstract data type); Order (exchange); Risk analysis (engineering); Work (physics); Knowledge management; Process management; Artificial intelligence; Engineering; Business; Cloud computing","score_opus":0.01849947260282962,"score_gpt":0.24430428134269927,"score_spread":0.22580480873986966,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2107479312","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025870038,0.0021490292,0.9591803,0.0015952466,0.00014727702,0.00010396234,0.000033169752,0.0003514342,0.010569578],"genre_scores_gemma":[0.8891368,0.0012247481,0.1053979,0.0003781134,0.0001869559,0.00025856862,0.000043781805,0.000050622308,0.0033225908],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985707,0.0007989573,0.000064721564,0.00017908395,0.00026871407,0.00011788207],"domain_scores_gemma":[0.99575305,0.003166108,0.00033888398,0.00018281121,0.0003691642,0.00018992726],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031116519,0.0005739242,0.0008577769,0.00044850237,0.00046104362,0.0013421777,0.0010759978,0.0011644496,0.0013693633],"category_scores_gemma":[0.0087788515,0.00032991893,0.00034522227,0.00049712486,0.0021656116,0.0013126363,0.00094976806,0.0017964058,0.00019811618],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007804482,0.000087989814,0.0008086861,0.00008295951,0.000051584026,0.000087796354,0.00010637015,0.89271545,0.00067076384,0.06816438,0.00073076255,0.036415305],"study_design_scores_gemma":[0.000028949396,0.000041070984,0.0001188051,0.000016521162,0.000008548014,0.00001523607,0.000014885852,0.94186175,0.00028917711,0.056479614,0.0011144252,0.000011034473],"about_ca_topic_score_codex":0.00446687,"about_ca_topic_score_gemma":0.002622532,"teacher_disagreement_score":0.00446687,"about_ca_system_score_codex":0.001654183,"about_ca_system_score_gemma":0.0016639275,"threshold_uncertainty_score":0.016456127},"labels":[],"label_agreement":null},{"id":"W2107741520","doi":"","title":"Weighted importance sampling for off-policy learning with linear function approximation","year":2014,"lang":"en","type":"article","venue":"neural information processing systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":100,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Weighting; Computer science; Importance sampling; Sampling (signal processing); Convergence (economics); Function (biology); Bridging (networking); Artificial intelligence; Mathematical optimization; Machine learning; Function approximation; Mathematics; Statistics; Artificial neural network","score_opus":0.020501250815071665,"score_gpt":0.2586979979424325,"score_spread":0.2381967471273608,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2107741520","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005933413,0.00014032076,0.9926208,0.00008012487,0.000029598914,0.000039601793,0.000011245715,0.00016012562,0.0009846879],"genre_scores_gemma":[0.69602567,0.00037897384,0.29873568,0.00026356356,0.00010210901,0.00041985002,0.0001569149,0.00018493328,0.00373229],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99833566,0.0007954943,0.00007223223,0.00020131363,0.00046619,0.00012920714],"domain_scores_gemma":[0.99480045,0.003931496,0.0002417963,0.00039393132,0.00048698587,0.00014535371],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003871777,0.0010368589,0.0015191423,0.00077235524,0.00046546655,0.0009972792,0.0015825137,0.0012385136,0.0024752496],"category_scores_gemma":[0.015067214,0.0006548598,0.00060754066,0.0007096122,0.0014912977,0.0015599575,0.001529916,0.0022186823,0.0004343321],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013791186,0.00011346711,0.0007565156,0.000093433555,0.000041269996,0.000071502574,0.00005880834,0.89446115,0.0011414099,0.044649154,0.0008608595,0.057614535],"study_design_scores_gemma":[0.000006589455,0.0000134382835,0.000024246418,0.000003260468,0.0000021753274,0.000005213354,0.000001674112,0.99230933,0.00018102884,0.0072921207,0.00015900865,0.0000017942385],"about_ca_topic_score_codex":0.003799199,"about_ca_topic_score_gemma":0.002948301,"teacher_disagreement_score":0.003871777,"about_ca_system_score_codex":0.0014544493,"about_ca_system_score_gemma":0.0017211516,"threshold_uncertainty_score":0.020476222},"labels":[],"label_agreement":null},{"id":"W2108005621","doi":"10.1609/aaai.v26i1.8260","title":"Sample Bounded Distributed Reinforcement Learning for Decentralized POMDPs","year":2021,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Reinforcement learning; Computer science; Partially observable Markov decision process; Benchmark (surveying); Markov decision process; Bounded function; Mathematical optimization; Sample (material); Sample complexity; Computation; Set (abstract data type); Artificial intelligence; Markov process; Markov chain; Machine learning; Mathematics; Markov model; Algorithm","score_opus":0.07689906790923026,"score_gpt":0.3054622507888089,"score_spread":0.22856318287957864,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2108005621","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027717398,0.00013945588,0.9699755,0.00030681392,0.000022654465,0.000057307407,0.000052185358,0.00026043208,0.001468381],"genre_scores_gemma":[0.9143212,0.00013240436,0.08374402,0.000090809204,0.000029030401,0.00026164416,0.00013110814,0.00007368232,0.0012162118],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99826366,0.0007433659,0.000084672334,0.00030403316,0.0004199192,0.00018421878],"domain_scores_gemma":[0.9868623,0.0105813835,0.0009039173,0.0006321298,0.0006269439,0.00039329214],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035040278,0.0010418231,0.0016011518,0.0004400761,0.00063109305,0.0011970932,0.0014219878,0.0010901889,0.0020136484],"category_scores_gemma":[0.016829347,0.00067604997,0.0006549965,0.00044599661,0.0019251857,0.0017343583,0.0019222092,0.002674219,0.00017909643],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000053516746,0.000027689317,0.00025467091,0.000034086162,0.000013161653,0.000024539355,0.000024812853,0.9836088,0.00024698835,0.011465318,0.00016242068,0.004084087],"study_design_scores_gemma":[0.0000093102735,0.000010318181,0.000024255392,0.00000195611,0.0000013343024,0.0000022326212,0.0000026901314,0.99159044,0.00008960245,0.008208821,0.000057622736,0.000001462334],"about_ca_topic_score_codex":0.0042704563,"about_ca_topic_score_gemma":0.004178923,"teacher_disagreement_score":0.0042704563,"about_ca_system_score_codex":0.002421401,"about_ca_system_score_gemma":0.0020887027,"threshold_uncertainty_score":0.018531263},"labels":[],"label_agreement":null},{"id":"W2109592579","doi":"10.1109/ccece.2003.1226121","title":"Using reinforcement learning for image thresholding","year":2004,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Reinforcement learning; Thresholding; Artificial intelligence; Image (mathematics); Machine learning; Computer vision","score_opus":0.05225257102805825,"score_gpt":0.30919521182997206,"score_spread":0.2569426408019138,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2109592579","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005916583,0.00017766173,0.9921428,0.000112268055,0.00003412719,0.000022452155,0.000004152411,0.00017513668,0.0014148807],"genre_scores_gemma":[0.7273843,0.0003753579,0.2691476,0.00019835356,0.00007577822,0.00013909054,0.000025011226,0.000079427315,0.0025750655],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994174,0.00022174392,0.000030318182,0.0001068315,0.00017977848,0.000043928187],"domain_scores_gemma":[0.9983298,0.0011027367,0.00016516105,0.000091844384,0.00023581529,0.00007465058],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012867183,0.00063013175,0.00078263885,0.00039314284,0.00027922186,0.0006964105,0.00095214276,0.0009258333,0.0014716821],"category_scores_gemma":[0.0047884043,0.00024172827,0.00037059622,0.00026315326,0.0011442014,0.0009584913,0.00089866685,0.0011870402,0.00026724857],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013431968,0.00012598999,0.0008598636,0.00016222346,0.00007146118,0.00012505997,0.0001249305,0.7772279,0.010851905,0.046104897,0.0012879737,0.16292349],"study_design_scores_gemma":[0.000014868133,0.00004526162,0.000077510755,0.000009958316,0.0000068214435,0.000026188196,0.0000052962264,0.9829333,0.0017625004,0.0144908205,0.0006182292,0.0000092156215],"about_ca_topic_score_codex":0.0015444196,"about_ca_topic_score_gemma":0.001152578,"teacher_disagreement_score":0.0015444196,"about_ca_system_score_codex":0.0007059739,"about_ca_system_score_gemma":0.00058572163,"threshold_uncertainty_score":0.0068048835},"labels":[],"label_agreement":null},{"id":"W2109909644","doi":"","title":"Cost-Sensitive Exploration in Bayesian Reinforcement Learning","year":2012,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Markov decision process; Computer science; ENCODE; Artificial intelligence; Total cost; Bayesian probability; Machine learning; Partially observable Markov decision process; Mathematical optimization; Bayesian optimization; Markov process; Markov chain; Markov model; Mathematics; Economics","score_opus":0.04822342128543226,"score_gpt":0.28527178582256785,"score_spread":0.2370483645371356,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2109909644","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04701057,0.00058159576,0.94754046,0.00059061404,0.000038940263,0.00005081113,0.00006423404,0.00017567402,0.0039470056],"genre_scores_gemma":[0.91748697,0.00034553232,0.07889765,0.00016101565,0.000043581054,0.00021270452,0.00007882168,0.000075541255,0.0026982077],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9983088,0.00095646107,0.000055914865,0.0001921697,0.00034279164,0.00014396166],"domain_scores_gemma":[0.9948526,0.0039652,0.00041275538,0.0002128191,0.0003331161,0.0002235005],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026310808,0.0010103177,0.001425913,0.0004970107,0.0003988618,0.0010192124,0.0013975286,0.0014391493,0.0020382504],"category_scores_gemma":[0.01216429,0.00066808,0.00051521143,0.0005451215,0.0018535224,0.0022323073,0.0015580462,0.0018006329,0.0001847842],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005572214,0.000030365863,0.00043090322,0.000061936254,0.000028677254,0.000049206657,0.00005669261,0.93211114,0.0004893808,0.05808942,0.0003247996,0.0082718395],"study_design_scores_gemma":[0.00001569763,0.000021676286,0.000058975096,0.000007036991,0.00000429594,0.000008274555,0.0000041957524,0.9678159,0.00010094584,0.031791247,0.00016628497,0.000005594563],"about_ca_topic_score_codex":0.004075246,"about_ca_topic_score_gemma":0.0031219132,"teacher_disagreement_score":0.004075246,"about_ca_system_score_codex":0.0017943803,"about_ca_system_score_gemma":0.0012665446,"threshold_uncertainty_score":0.013914645},"labels":[],"label_agreement":null},{"id":"W2111625536","doi":"","title":"Value Pursuit Iteration","year":2012,"lang":"en","type":"article","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Bellman equation; Reinforcement learning; Value (mathematics); Function (biology); Set (abstract data type); Mathematics; Algorithm; Power iteration; Mathematical optimization; Representation (politics); Approximation error; Computer science; Iterative method; Approximation algorithm; Artificial intelligence; Statistics","score_opus":0.013020207704836174,"score_gpt":0.23781960109386333,"score_spread":0.22479939338902716,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2111625536","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032729188,0.0003000762,0.98972017,0.00018411361,0.000060986044,0.00006701226,0.000024839863,0.00021301654,0.006156876],"genre_scores_gemma":[0.40786445,0.0010744343,0.57170445,0.0005185254,0.00019998687,0.0006414736,0.0003304526,0.00042153487,0.017244674],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977532,0.0007713995,0.00011562619,0.00039879186,0.0007034642,0.00025753689],"domain_scores_gemma":[0.99485624,0.0033291248,0.0002842499,0.00045039083,0.00085639965,0.00022355972],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027775036,0.0015028451,0.0021796145,0.0007773086,0.0008517262,0.0019831492,0.001708583,0.0019636154,0.0066468087],"category_scores_gemma":[0.014039382,0.00066628086,0.0008850553,0.0008346173,0.0022125526,0.002058533,0.0033157414,0.0029392568,0.00195151],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032092078,0.00017299266,0.0011174999,0.00032141898,0.0001289985,0.00017707808,0.00017861073,0.5211072,0.0030560493,0.2733959,0.008300462,0.19172294],"study_design_scores_gemma":[0.000032904536,0.000070722686,0.0000622089,0.000031890075,0.00000988718,0.0000437926,0.000015883697,0.94178075,0.0011960545,0.053674363,0.0030700164,0.000011539172],"about_ca_topic_score_codex":0.0017118255,"about_ca_topic_score_gemma":0.0014238268,"teacher_disagreement_score":0.0066468087,"about_ca_system_score_codex":0.0014175923,"about_ca_system_score_gemma":0.0023837357,"threshold_uncertainty_score":0.02223581},"labels":[],"label_agreement":null},{"id":"W2111680853","doi":"","title":"A planning algorithm for predictive state representations","year":2003,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Partially observable Markov decision process; Observable; Representation (politics); Markov decision process; Computer science; State (computer science); Sequence (biology); Mathematical optimization; Markov process; Process (computing); Algorithm; Mathematics; Statistics","score_opus":0.027297718444463843,"score_gpt":0.3026270110575934,"score_spread":0.27532929261312955,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2111680853","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013944538,0.00007211813,0.99600226,0.00009925108,0.000021982914,0.00004167239,0.000027180922,0.00041445525,0.001926649],"genre_scores_gemma":[0.1240016,0.00025081006,0.87140554,0.00011873908,0.000047509697,0.00041670882,0.00018868671,0.00017280421,0.003397642],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999252,0.00020673113,0.00003878862,0.00020986974,0.00021157166,0.00008094951],"domain_scores_gemma":[0.9990877,0.0006115537,0.000054017804,0.00010756676,0.000097583565,0.00004152304],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012114686,0.0010045575,0.000941233,0.00073002215,0.00070362637,0.001177961,0.0019254871,0.0015965459,0.005580661],"category_scores_gemma":[0.0042063864,0.00063094386,0.000978425,0.0008967304,0.0011952496,0.0022594205,0.0019659868,0.0023135757,0.0010944757],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006385968,0.00008509084,0.0002805657,0.00010902118,0.0000370408,0.00011377642,0.00015641424,0.64838374,0.0013715872,0.17978162,0.004179731,0.1654376],"study_design_scores_gemma":[0.000016876576,0.00001744514,0.000018486497,0.000011777592,0.000007790128,0.000018980238,0.000009715675,0.955848,0.00052835635,0.040935256,0.0025814031,0.000005920473],"about_ca_topic_score_codex":0.003704174,"about_ca_topic_score_gemma":0.0032652945,"teacher_disagreement_score":0.005580661,"about_ca_system_score_codex":0.001006133,"about_ca_system_score_gemma":0.0019394562,"threshold_uncertainty_score":0.018669188},"labels":[],"label_agreement":null},{"id":"W2111833414","doi":"10.1109/robot.2008.4543641","title":"Bayesian reinforcement learning in continuous POMDPs with application to robot navigation","year":2008,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":68,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval; McGill University","funders":"","keywords":"Computer science; Reinforcement learning; Markov decision process; Particle filter; Partially observable Markov decision process; Robot; Artificial intelligence; Trajectory; Optimal control; Posterior probability; Bayesian probability; Mathematical optimization; Observable; Machine learning; Markov process; Markov chain; Kalman filter; Markov model; Mathematics","score_opus":0.009732780240815567,"score_gpt":0.23108885917368907,"score_spread":0.2213560789328735,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2111833414","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01375947,0.0006105661,0.9834155,0.00034278768,0.000043584427,0.000026009795,0.000028704379,0.00022345009,0.0015499006],"genre_scores_gemma":[0.78651786,0.0008696203,0.2095966,0.00014939824,0.000107480955,0.00021158633,0.00010174354,0.00007818347,0.0023675226],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99918467,0.00038285894,0.000034352277,0.00010863394,0.00021240082,0.00007714519],"domain_scores_gemma":[0.9957581,0.0033727877,0.0002768066,0.00011029269,0.00032121397,0.00016075799],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020120677,0.00089517835,0.0014621114,0.00054447167,0.00050138234,0.00086316734,0.0012203546,0.001228384,0.0016536673],"category_scores_gemma":[0.009687849,0.0006176473,0.000598077,0.0008107098,0.0015412391,0.0013241167,0.0011184397,0.0018174717,0.00019477987],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000038252958,0.000032559645,0.0003392447,0.00004900917,0.000021078802,0.000046419802,0.000045470326,0.96864533,0.00023513005,0.016477415,0.0002701776,0.013799816],"study_design_scores_gemma":[0.000011807898,0.000010172182,0.00004211904,0.0000038654784,0.0000030963756,0.00000420301,0.0000029674563,0.9901438,0.00006667147,0.009556461,0.00015139094,0.0000034350849],"about_ca_topic_score_codex":0.013711153,"about_ca_topic_score_gemma":0.0075309398,"teacher_disagreement_score":0.013711153,"about_ca_system_score_codex":0.0013861359,"about_ca_system_score_gemma":0.0013850544,"threshold_uncertainty_score":0.027262688},"labels":[],"label_agreement":null},{"id":"W2111836691","doi":"10.1109/iros.2009.5354126","title":"Robot task switching under diminishing returns","year":2009,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Robot; Work (physics); Computer science; Term (time); Rate of return; Task (project management); Foraging; Econometrics; Mathematical optimization; Variety (cybernetics); Economics; Mathematics; Artificial intelligence; Engineering; Finance","score_opus":0.015211187821304775,"score_gpt":0.25029447934549937,"score_spread":0.2350832915241946,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2111836691","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5157179,0.000426624,0.4729908,0.0008927873,0.000043322558,0.00010964042,0.00010672285,0.00056877366,0.009143358],"genre_scores_gemma":[0.9808677,0.00008759597,0.01629749,0.00006727589,0.000015496416,0.000100364174,0.000038840313,0.000047529462,0.0024777306],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99924797,0.0003218906,0.000029060915,0.00013762199,0.00011428714,0.00014922323],"domain_scores_gemma":[0.99529403,0.0030716162,0.00051637006,0.00037258668,0.0002920701,0.00045326786],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020191108,0.0006171333,0.0012641684,0.00055423577,0.00041276624,0.0007699578,0.0016635583,0.0010880989,0.0027840382],"category_scores_gemma":[0.0111346375,0.0003935512,0.000503172,0.0004136543,0.0012312206,0.0015913016,0.0012897581,0.0008558322,0.00048591412],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024666724,0.00011143555,0.0013185856,0.000065173204,0.000051115632,0.00014710586,0.00017436482,0.9412317,0.0025537596,0.025348227,0.0008051905,0.027946703],"study_design_scores_gemma":[0.000042171752,0.000075050346,0.00034840842,0.000004242717,0.0000099768,0.000030526648,0.000025360732,0.9712701,0.0005769002,0.02731512,0.0002922873,0.000009978665],"about_ca_topic_score_codex":0.0018307972,"about_ca_topic_score_gemma":0.0011279855,"teacher_disagreement_score":0.0027840382,"about_ca_system_score_codex":0.0009913732,"about_ca_system_score_gemma":0.0006307043,"threshold_uncertainty_score":0.010678172},"labels":[],"label_agreement":null},{"id":"W2113710263","doi":"10.1109/icca.2003.1595047","title":"A Reinfrocement Learning Approach to Online Learning in Control","year":2003,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Control (management); Online learning; Artificial intelligence; Multimedia","score_opus":0.01819162284227271,"score_gpt":0.24589915820550085,"score_spread":0.22770753536322813,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2113710263","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0023688786,0.0002611912,0.99508035,0.0001782656,0.000027765087,0.0000123579875,0.0000047333065,0.00008665465,0.0019797971],"genre_scores_gemma":[0.55296564,0.0009701622,0.42978364,0.00041557578,0.00023563232,0.00019377474,0.000039593393,0.00010208909,0.015293903],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99957687,0.00017082151,0.000017207614,0.000073371106,0.00012805928,0.00003365745],"domain_scores_gemma":[0.99947673,0.00029385832,0.000037688547,0.0000754682,0.00009196324,0.000024387222],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010053815,0.0005203913,0.00070457417,0.00035693304,0.00038077537,0.00061830133,0.0014914484,0.0009778626,0.003436786],"category_scores_gemma":[0.002040972,0.000250868,0.00055709766,0.00039284184,0.0014940746,0.0013092946,0.00082862814,0.0017518846,0.00043790822],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007011652,0.00006784088,0.00028639517,0.00014318016,0.000058270834,0.00012416135,0.0001909851,0.5020033,0.0044554425,0.35402814,0.0016520033,0.1369202],"study_design_scores_gemma":[0.000014363082,0.000086104075,0.00004851375,0.000009846883,0.0000054621237,0.000032283642,0.0000079817355,0.9317622,0.00096797134,0.06385331,0.003203661,0.000008289029],"about_ca_topic_score_codex":0.0019970825,"about_ca_topic_score_gemma":0.0016533376,"teacher_disagreement_score":0.003436786,"about_ca_system_score_codex":0.0007076143,"about_ca_system_score_gemma":0.00057282136,"threshold_uncertainty_score":0.01149714},"labels":[],"label_agreement":null},{"id":"W2114500662","doi":"10.1142/s0129183101002851","title":"DEEP-SARSA: A REINFORCEMENT LEARNING ALGORITHM FOR AUTONOMOUS NAVIGATION","year":2001,"lang":"en","type":"article","venue":"International Journal of Modern Physics C","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Reinforcement learning; Computer science; Convergence (economics); Algorithm; Artificial intelligence; Q-learning; Learning classifier system; Robot; Graph; Theoretical computer science","score_opus":0.019940586004140722,"score_gpt":0.2823315718161346,"score_spread":0.2623909858119939,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2114500662","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0040758164,0.0001136481,0.99457645,0.000077663746,0.00003861788,0.000021787682,0.000015248613,0.00038307506,0.00069765415],"genre_scores_gemma":[0.39969182,0.00024174125,0.59527904,0.00018479886,0.000044166638,0.00021790247,0.00010938564,0.0001250581,0.0041061305],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997813,0.000072420895,0.000012440381,0.000045973593,0.000063693566,0.00002423262],"domain_scores_gemma":[0.9993783,0.0003350431,0.000058314672,0.000051693998,0.00013214038,0.00004457806],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00082898507,0.0005813434,0.0006794973,0.00034138944,0.0002773577,0.00044908246,0.0010156744,0.00096084294,0.0024059],"category_scores_gemma":[0.0018505757,0.0003097966,0.00035262146,0.00027323904,0.00075736723,0.0008278961,0.0008182122,0.0013574706,0.00052200136],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013201886,0.000075546464,0.0004919454,0.000094701565,0.000044573648,0.000049824896,0.00005191534,0.8124688,0.004831839,0.031618554,0.002290477,0.1478498],"study_design_scores_gemma":[0.000014527949,0.000031763142,0.000029382803,0.0000037692462,0.000002872785,0.000011148413,0.0000023801215,0.99250716,0.0006942292,0.0057381564,0.0009609383,0.0000036742533],"about_ca_topic_score_codex":0.0023308175,"about_ca_topic_score_gemma":0.0022285616,"teacher_disagreement_score":0.0024059,"about_ca_system_score_codex":0.000561672,"about_ca_system_score_gemma":0.0011626662,"threshold_uncertainty_score":0.008048594},"labels":[],"label_agreement":null},{"id":"W2114819182","doi":"","title":"Optimal Robot Recharging Strategies For Time Discounted Labour","year":2008,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Work (physics); Robot; Computer science; Energy (signal processing); Selection (genetic algorithm); Value (mathematics); Action (physics); Class (philosophy); Simulation; Artificial intelligence; Mathematics; Engineering; Machine learning","score_opus":0.025938479480937058,"score_gpt":0.26250877982827586,"score_spread":0.2365703003473388,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2114819182","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.45097294,0.0005394527,0.5293252,0.0013675164,0.00006227933,0.00016148237,0.00010649196,0.0002999623,0.017164636],"genre_scores_gemma":[0.98630965,0.00006992807,0.011040344,0.00005178852,0.0000065481527,0.000051445124,0.000018844774,0.000021846772,0.0024294702],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99956447,0.00017414398,0.00001864221,0.00007220928,0.00006283596,0.00010781625],"domain_scores_gemma":[0.9986645,0.00071375334,0.00025272847,0.000088343535,0.00011182355,0.00016891633],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010832595,0.0008478098,0.0010910747,0.0006187456,0.00040778008,0.0010569199,0.0013719964,0.0015096795,0.0029947176],"category_scores_gemma":[0.003809011,0.0005313805,0.00038396177,0.00038559444,0.001402063,0.0015992792,0.0010799376,0.0007614832,0.0003267836],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010569344,0.000046064746,0.00040397813,0.00003474998,0.000019324183,0.0000771129,0.00007529928,0.9787911,0.0008805561,0.012405037,0.0003310956,0.0068299253],"study_design_scores_gemma":[0.000024909177,0.000054452947,0.00013382164,0.000007712515,0.0000069575176,0.000014434815,0.000041616277,0.9875421,0.00021794894,0.011672854,0.00027515483,0.00000806683],"about_ca_topic_score_codex":0.0042217337,"about_ca_topic_score_gemma":0.0029412152,"teacher_disagreement_score":0.0042217337,"about_ca_system_score_codex":0.0015524109,"about_ca_system_score_gemma":0.00088694214,"threshold_uncertainty_score":0.011263609},"labels":[],"label_agreement":null},{"id":"W2115083386","doi":"10.1007/978-3-642-04174-7_43","title":"Learning the Difference between Partially Observable Dynamical Systems","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval; McGill University","funders":"","keywords":"Observable; Computer science; Markov decision process; Divergence (linguistics); Reinforcement learning; Dynamical systems theory; Key (lock); Temporal difference learning; Markov process; Markov chain; Artificial intelligence; Process (computing); Mathematical optimization; Machine learning; Mathematics; Statistics","score_opus":0.025984326590315347,"score_gpt":0.24446419850116202,"score_spread":0.21847987191084667,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2115083386","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07526908,0.00040296628,0.91992664,0.00042758632,0.00013687584,0.000032609696,0.000089739086,0.00035460794,0.0033598158],"genre_scores_gemma":[0.90827775,0.00024335111,0.08763633,0.000115745155,0.00007163843,0.000081640625,0.00024520454,0.000055981218,0.0032723416],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99930847,0.0002287291,0.000041336654,0.00020071842,0.00017035028,0.000050377635],"domain_scores_gemma":[0.9952094,0.003888085,0.0002211517,0.00030373377,0.00022132746,0.00015622623],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015077788,0.0004567291,0.0008455677,0.0003645122,0.00019346812,0.0008867414,0.0012036219,0.0008923366,0.0028486603],"category_scores_gemma":[0.00962594,0.0004553697,0.00039868202,0.00026918488,0.0011309098,0.0026883804,0.001929607,0.0022464243,0.0002730841],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042921584,0.000134629,0.0025479149,0.00031549082,0.00010198163,0.000087096014,0.00019641426,0.5602937,0.005422303,0.19682899,0.0016406063,0.23200166],"study_design_scores_gemma":[0.000019749557,0.00007824954,0.00024277467,0.000011435094,0.000006387787,0.000017142225,0.00000879917,0.92143553,0.0006715333,0.07705445,0.00044469934,0.000009319168],"about_ca_topic_score_codex":0.00069306593,"about_ca_topic_score_gemma":0.0007601373,"teacher_disagreement_score":0.0028486603,"about_ca_system_score_codex":0.00059249956,"about_ca_system_score_gemma":0.00044110362,"threshold_uncertainty_score":0.00952971},"labels":[],"label_agreement":null},{"id":"W2115253045","doi":"","title":"Off-policy Learning with Options and Recognizers","year":2005,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; McGill University","funders":"","keywords":"Variance (accounting); Computer science; Convergence (economics); Function (biology); Temporal difference learning; Reinforcement learning; Sampling (signal processing); Filter (signal processing); Function approximation; State (computer science); Importance sampling; Artificial intelligence; Policy learning; Algorithm; Machine learning; Mathematical optimization; Mathematics; Statistics; Artificial neural network","score_opus":0.012605854074279685,"score_gpt":0.24240306836910663,"score_spread":0.22979721429482694,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2115253045","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007782258,0.00005790996,0.9906855,0.00008848803,0.000027935886,0.000034262186,0.000009168648,0.0002673907,0.0010469491],"genre_scores_gemma":[0.5298639,0.0001256404,0.46499062,0.00028932327,0.00006363437,0.00024337452,0.00008311288,0.00014541678,0.0041949553],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99870837,0.00038258758,0.00008652303,0.00026880906,0.00043684433,0.00011688849],"domain_scores_gemma":[0.9965228,0.0021713166,0.0002860278,0.0005093619,0.00036592936,0.00014461714],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024591382,0.00067106093,0.0011321504,0.00059713784,0.00033419958,0.0012035273,0.0021970554,0.0014962364,0.0026611062],"category_scores_gemma":[0.009931841,0.0005324236,0.0006099107,0.00050325907,0.0016785196,0.0024054593,0.001991589,0.0021480783,0.0005040884],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028150986,0.00020674465,0.0011882557,0.00009420468,0.000059712766,0.00014461196,0.00018132315,0.56944543,0.0048804837,0.12621345,0.0014038198,0.29590052],"study_design_scores_gemma":[0.000011954947,0.000031036227,0.00004314151,0.0000050310873,0.0000036151923,0.000019744917,0.0000042806496,0.9863013,0.0010489413,0.012085716,0.0004403154,0.0000049222162],"about_ca_topic_score_codex":0.0019423595,"about_ca_topic_score_gemma":0.0014436734,"teacher_disagreement_score":0.0026611062,"about_ca_system_score_codex":0.0011977518,"about_ca_system_score_gemma":0.0012763464,"threshold_uncertainty_score":0.013005316},"labels":[],"label_agreement":null},{"id":"W2115318338","doi":"","title":"Bootstrapping Apprenticeship Learning","year":2010,"lang":"en","type":"article","venue":"Max Planck Digital Library","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Bootstrapping (finance); Reinforcement learning; Feature (linguistics); Machine learning; Artificial intelligence; Monte Carlo method; Quality (philosophy); State space; Feature engineering; Feature vector; Apprenticeship; Mathematical optimization; Mathematics; Econometrics; Statistics","score_opus":0.006204349322123965,"score_gpt":0.18233502947085264,"score_spread":0.17613068014872868,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2115318338","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.092265874,0.00028740344,0.8986707,0.0005763285,0.00010464792,0.0002605011,0.000062490515,0.0013670542,0.006405104],"genre_scores_gemma":[0.8711345,0.00009849542,0.12308272,0.00030462552,0.000043949716,0.0003134103,0.00019385791,0.00010693983,0.0047215573],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987488,0.000504072,0.00005010083,0.0003506086,0.00020544807,0.00014092916],"domain_scores_gemma":[0.9934854,0.0040568747,0.00033450214,0.0009095174,0.0007021085,0.0005117026],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016582562,0.00091756653,0.0014249722,0.00079068757,0.0006478587,0.0006769165,0.0033745945,0.0016288831,0.005883829],"category_scores_gemma":[0.011335468,0.00044065816,0.0007343664,0.0005274176,0.0015235731,0.0017247661,0.003129226,0.0023294818,0.0010664271],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038745045,0.000755258,0.005862844,0.00017812244,0.00011002623,0.0004559617,0.00034433964,0.5394905,0.0025998913,0.04513932,0.005484449,0.39919177],"study_design_scores_gemma":[0.00004047171,0.00011840607,0.00021243813,0.00001554476,0.00001006227,0.00007087173,0.00002281946,0.97309905,0.0011521832,0.0242435,0.0010066609,0.000007984307],"about_ca_topic_score_codex":0.0014761918,"about_ca_topic_score_gemma":0.0017129692,"teacher_disagreement_score":0.005883829,"about_ca_system_score_codex":0.0006615827,"about_ca_system_score_gemma":0.0009968599,"threshold_uncertainty_score":0.01968342},"labels":[],"label_agreement":null},{"id":"W2115615930","doi":"10.1109/ijcnn.2006.247204","title":"Extend Single-agent Reinforcement Learning Approach to a Multi-robot Cooperative Task in an Unknown Dynamic Environment","year":2006,"lang":"en","type":"article","venue":"The 2006 IEEE International Joint Conference on Neural Network Proceedings","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Reinforcement learning; Computer science; Robot; Markov decision process; Robustness (evolution); Robot learning; Artificial intelligence; Q-learning; Obstacle; Mobile robot; Markov process; Task (project management); Machine learning; Engineering; Mathematics","score_opus":0.052808010837956824,"score_gpt":0.26725047299880855,"score_spread":0.21444246216085172,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2115615930","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016313445,0.00011915279,0.9818392,0.00017255326,0.000033178003,0.00003625384,0.000009320058,0.00015281467,0.0013240209],"genre_scores_gemma":[0.8544557,0.00023207067,0.14244008,0.0001464443,0.000041930358,0.00014223858,0.000030434083,0.000028778053,0.002482333],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996356,0.00013504444,0.00002068958,0.00008474714,0.00008204321,0.000041871135],"domain_scores_gemma":[0.99926907,0.00041149085,0.00007389131,0.000075684424,0.00011631343,0.000053640255],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009824346,0.0006061138,0.0007484176,0.00021911402,0.0003270694,0.00040712536,0.0009693963,0.0007411573,0.0013827368],"category_scores_gemma":[0.001574774,0.0002174507,0.00057565817,0.00022235679,0.0007291179,0.0008762276,0.00076441653,0.0009221503,0.00025111233],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000046619578,0.00008590785,0.00074913265,0.0000818676,0.000050052106,0.00022339486,0.00010132473,0.9375379,0.0040827133,0.014020315,0.00040040957,0.042620383],"study_design_scores_gemma":[0.000009644513,0.000035189852,0.000057993926,0.0000022162692,0.0000048472766,0.000019644414,0.000004276373,0.99501,0.00045176139,0.004001485,0.00039938354,0.0000034290551],"about_ca_topic_score_codex":0.0021157216,"about_ca_topic_score_gemma":0.0012105989,"teacher_disagreement_score":0.0021157216,"about_ca_system_score_codex":0.0004123751,"about_ca_system_score_gemma":0.00083136064,"threshold_uncertainty_score":0.0051956773},"labels":[],"label_agreement":null},{"id":"W2117723247","doi":"","title":"An on-line decision-theoretic Golog interpreter","year":2001,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Interpreter; Computer science; Prolog; Programming language; Representation (politics); Artificial intelligence","score_opus":0.019679629969391682,"score_gpt":0.2986808653297143,"score_spread":0.2790012353603226,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2117723247","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020413885,0.00009936169,0.96804124,0.0003622879,0.00004021814,0.00018910335,0.00016552159,0.0035758393,0.007112511],"genre_scores_gemma":[0.23407693,0.00016300769,0.75998753,0.00023938925,0.000029203908,0.00023211366,0.00022994126,0.00042320427,0.0046186345],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992567,0.0002228126,0.000055505723,0.00013849745,0.00020965887,0.00011689613],"domain_scores_gemma":[0.9990081,0.0005298622,0.0001049945,0.0001378138,0.00013759815,0.0000815939],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012057034,0.0005520081,0.0004420501,0.0003834941,0.00045204585,0.0016634634,0.0018979829,0.0008594766,0.0051237145],"category_scores_gemma":[0.0037621295,0.00040641383,0.0007916205,0.00031575825,0.0019677277,0.0018179873,0.0013425171,0.0013465823,0.0009458375],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045904014,0.00040467753,0.0021163668,0.0006297354,0.00008223852,0.0010141514,0.0011547848,0.35877195,0.01849811,0.43810886,0.0068979524,0.1718622],"study_design_scores_gemma":[0.000103583334,0.00013130737,0.00022645529,0.00007755636,0.000042807904,0.00022753686,0.00011974658,0.84392756,0.013228803,0.12506577,0.016815873,0.000032984466],"about_ca_topic_score_codex":0.0026217888,"about_ca_topic_score_gemma":0.0030451862,"teacher_disagreement_score":0.0051237145,"about_ca_system_score_codex":0.0010030976,"about_ca_system_score_gemma":0.0017066686,"threshold_uncertainty_score":0.017140508},"labels":[],"label_agreement":null},{"id":"W2119076930","doi":"10.1007/978-3-540-30217-9_100","title":"A Neuroevolutionary Approach to Emergent Task Decomposition","year":2004,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Decomposition; Task (project management); Modular design; Artificial neural network; Artificial intelligence; Scalability; Reinforcement learning; Exploit; Function (biology); Distributed computing; Programming language","score_opus":0.016081137430461346,"score_gpt":0.24822591734682667,"score_spread":0.2321447799163653,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2119076930","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022936355,0.00038492083,0.96130425,0.00048170102,0.00007315671,0.00003099364,0.000045904326,0.00021200598,0.014530675],"genre_scores_gemma":[0.51402,0.00066771766,0.46718496,0.00016500453,0.00008433918,0.00021282687,0.0001432611,0.0002693631,0.017252546],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99987614,0.000034649296,0.000006746385,0.000029002025,0.000034743007,0.000018668023],"domain_scores_gemma":[0.9997398,0.0001292839,0.000020532358,0.000037366946,0.00004546523,0.00002759231],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00034361548,0.00044379532,0.00050032727,0.00039086404,0.0005186363,0.0007947796,0.001261619,0.001019277,0.0048129396],"category_scores_gemma":[0.0015754985,0.0004224235,0.0006324976,0.00042116822,0.001042447,0.0010307069,0.0012321444,0.0014952373,0.0003502023],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000021330614,0.00002840887,0.0003421854,0.000047948037,0.000048750975,0.00007890652,0.00021809658,0.666089,0.004353404,0.25531518,0.0023674285,0.0710894],"study_design_scores_gemma":[0.000005801056,0.000012047128,0.00012751398,0.000006640999,0.0000068985623,0.00002342787,0.00002118737,0.8863809,0.00030381727,0.11160821,0.0014959844,0.0000075032235],"about_ca_topic_score_codex":0.0035501397,"about_ca_topic_score_gemma":0.0037063195,"teacher_disagreement_score":0.0048129396,"about_ca_system_score_codex":0.0007971484,"about_ca_system_score_gemma":0.00050649664,"threshold_uncertainty_score":0.016100824},"labels":[],"label_agreement":null},{"id":"W2119778140","doi":"10.11575/prism/31038","title":"Spidey: a Robotic Tabletop Assistant","year":2012,"lang":"en","type":"article","venue":"Open MIND","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Human–computer interaction; Computer science; Visualization; Robot; Task (project management); Rapid prototyping; Artificial intelligence; Systems engineering; Engineering","score_opus":0.06420558674325796,"score_gpt":0.31618562482708984,"score_spread":0.25198003808383185,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2119778140","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.059518848,0.00041233422,0.9140431,0.0003340686,0.0001325206,0.0005289888,0.00016068686,0.009799098,0.015070243],"genre_scores_gemma":[0.37464696,0.00059001305,0.5975482,0.00029171383,0.00006719618,0.0004945962,0.00032649186,0.00038819632,0.025646672],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9995771,0.00006167495,0.000023232735,0.00008547927,0.00020740058,0.00004509839],"domain_scores_gemma":[0.9993783,0.00022675189,0.000056978264,0.00012074753,0.00008964324,0.00012766599],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000399443,0.00068836,0.0005189686,0.00029959693,0.00035292594,0.0009591139,0.0018261186,0.0008171709,0.011209133],"category_scores_gemma":[0.0014974674,0.00039182956,0.00046656735,0.00015485813,0.00071814505,0.0014817386,0.0021265543,0.0010196986,0.0020932297],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015797254,0.0005156754,0.0019474198,0.0012925623,0.00013153504,0.0015609071,0.0018073672,0.036131363,0.3391639,0.022164404,0.014762428,0.57894284],"study_design_scores_gemma":[0.00073421793,0.004263866,0.005228634,0.00027925926,0.00018853128,0.0043369993,0.00084504334,0.38941938,0.18122645,0.01198651,0.40115607,0.0003350881],"about_ca_topic_score_codex":0.00070514274,"about_ca_topic_score_gemma":0.0009229128,"teacher_disagreement_score":0.011209133,"about_ca_system_score_codex":0.0001922394,"about_ca_system_score_gemma":0.00056365336,"threshold_uncertainty_score":0.037498295},"labels":[],"label_agreement":null},{"id":"W2119785746","doi":"10.1007/s10994-009-5110-1","title":"Training parsers by inverse reinforcement learning","year":2009,"lang":"en","type":"article","venue":"Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":70,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Treebank; Parsing; Computer science; Generalization; Artificial intelligence; Set (abstract data type); Reinforcement learning; Function (biology); Machine learning; Training set; Inverse; Algorithm; Mathematics","score_opus":0.02311423201309188,"score_gpt":0.2528785846531024,"score_spread":0.22976435264001055,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2119785746","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025625339,0.0002768507,0.9664972,0.00061458076,0.00012930864,0.000086171734,0.000097939235,0.004485401,0.0021871421],"genre_scores_gemma":[0.6120088,0.00023312241,0.38096702,0.00053989363,0.000083706655,0.0003628181,0.0004070113,0.0006009729,0.0047966433],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99874926,0.00049974327,0.000070638846,0.00035604704,0.00019279686,0.00013161992],"domain_scores_gemma":[0.98973197,0.008226462,0.000307519,0.0007401834,0.00080433703,0.000189618],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025821405,0.001040894,0.0015875855,0.00095202023,0.0006153601,0.0011036715,0.0021877924,0.0024038432,0.006075742],"category_scores_gemma":[0.014184031,0.0014768351,0.0010232447,0.000593291,0.0017775021,0.002852054,0.001926974,0.004279763,0.0014222547],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030707335,0.0002960386,0.0022561727,0.00024763844,0.00013449905,0.00024073185,0.00023736226,0.6017127,0.005326719,0.029867139,0.0082284175,0.35114548],"study_design_scores_gemma":[0.000025189664,0.000022960989,0.00006721588,0.000012522282,0.000014455677,0.000019525394,0.000009188798,0.9829047,0.0012002994,0.015259382,0.0004571209,0.000007382752],"about_ca_topic_score_codex":0.004106336,"about_ca_topic_score_gemma":0.005273471,"teacher_disagreement_score":0.006075742,"about_ca_system_score_codex":0.0012214467,"about_ca_system_score_gemma":0.0022697484,"threshold_uncertainty_score":0.020325363},"labels":[],"label_agreement":null},{"id":"W2120045945","doi":"10.1287/moor.2016.0832","title":"On the Asymptotic Optimality of Finite Approximations to Markov Decision Processes with Borel Spaces","year":2017,"lang":"en","type":"preprint","venue":"Mathematics of Operations Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Markov decision process; Mathematics; State space; Convergence (economics); Applied mathematics; Mathematical optimization; Action (physics); Markov chain; Space (punctuation); Finite state; Markov process; Class (philosophy); Q-learning; State (computer science); Average cost; Reinforcement learning; Computer science; Algorithm; Statistics","score_opus":0.11485889613744056,"score_gpt":0.3944555419249472,"score_spread":0.27959664578750665,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2120045945","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04024952,0.0011756549,0.9530747,0.0007299379,0.00006513998,0.0000537685,0.000081391365,0.00021933187,0.004350571],"genre_scores_gemma":[0.80245847,0.0022833273,0.18992405,0.00039815877,0.00021019284,0.00041358118,0.00048982725,0.00035069027,0.003471743],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99408144,0.0031439562,0.00029584926,0.00072669366,0.001284133,0.00046786922],"domain_scores_gemma":[0.8984431,0.092752084,0.0028169623,0.0027166838,0.0023792977,0.0008919176],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0120010665,0.0016404517,0.002271788,0.0023398653,0.0011424059,0.003032789,0.002386183,0.0021784077,0.0024164785],"category_scores_gemma":[0.09989255,0.0010752559,0.001665893,0.0013456342,0.005724216,0.0052773026,0.0034113277,0.0053998325,0.00046157616],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012161009,0.000081165526,0.0013514675,0.00018310064,0.000082093386,0.00009271746,0.00020201974,0.7005761,0.0008023205,0.28259218,0.0005926023,0.013322633],"study_design_scores_gemma":[0.000008217061,0.000030805062,0.00012452052,0.000038701124,0.0000074578124,0.00001612549,0.000017449118,0.8959663,0.0003265594,0.10323824,0.00021651355,0.000009009146],"about_ca_topic_score_codex":0.0044772,"about_ca_topic_score_gemma":0.0028455185,"teacher_disagreement_score":0.0120010665,"about_ca_system_score_codex":0.004066811,"about_ca_system_score_gemma":0.0028371597,"threshold_uncertainty_score":0.063468456},"labels":[],"label_agreement":null},{"id":"W2121943493","doi":"","title":"Exact Dynamic Programming for decentralized POMDPs with lossless policy compression","year":2008,"lang":"en","type":"article","venue":"MPG.PuRe (Max Planck Society)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Benchmark (surveying); Curse of dimensionality; Dynamic programming; Mathematical optimization; State space; Reinforcement learning; Lossless compression; Partially observable Markov decision process; Markov decision process; Computation; Compression (physics); State (computer science); Theoretical computer science; Artificial intelligence; Data compression; Algorithm; Machine learning; Mathematics; Markov process; Markov chain; Markov model","score_opus":0.016323954191927535,"score_gpt":0.26402618162316643,"score_spread":0.2477022274312389,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2121943493","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008836369,0.00017986307,0.9880634,0.00021062826,0.000038546743,0.000044144184,0.00009538527,0.0002627613,0.0022688434],"genre_scores_gemma":[0.6910134,0.00040567774,0.3007345,0.00022227527,0.000104845305,0.0004934812,0.0003791094,0.00016714112,0.0064795287],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999453,0.00016344749,0.00003185645,0.00009983488,0.00017837812,0.0000734971],"domain_scores_gemma":[0.99858826,0.0010117901,0.00009985662,0.00015103271,0.00010077205,0.000048228085],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011834234,0.00083637435,0.0014106138,0.00045974745,0.00037051053,0.0010463543,0.0010157562,0.001069408,0.003209513],"category_scores_gemma":[0.0045830654,0.0005925569,0.0005663967,0.0007891633,0.0008679801,0.0019292864,0.0018346214,0.0019204691,0.00042358],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000049852704,0.000026302543,0.00010014239,0.000053448908,0.000014910638,0.00002102352,0.00002276608,0.94945663,0.00033715158,0.023669504,0.0007467049,0.025501562],"study_design_scores_gemma":[0.00000778355,0.000008540303,0.0000145740305,0.0000030318238,0.0000017195224,0.0000032858327,0.0000018899242,0.9863067,0.00012030573,0.013364551,0.0001659032,0.0000016195096],"about_ca_topic_score_codex":0.0025583801,"about_ca_topic_score_gemma":0.0027215765,"teacher_disagreement_score":0.003209513,"about_ca_system_score_codex":0.0009969026,"about_ca_system_score_gemma":0.0013991406,"threshold_uncertainty_score":0.010736823},"labels":[],"label_agreement":null},{"id":"W2122372591","doi":"","title":"Sketch-Based Linear Value Function Approximation","year":2012,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Hash function; Computer science; Estimator; Reinforcement learning; Universal hashing; Function approximation; Dynamic perfect hashing; Function (biology); Bellman equation; Hash table; Artificial intelligence; Algorithm; Mathematics; Mathematical optimization; Statistics; Double hashing; Artificial neural network","score_opus":0.02141704514927115,"score_gpt":0.24769549645635106,"score_spread":0.2262784513070799,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2122372591","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010098439,0.00049351907,0.9871587,0.0001563446,0.000030755175,0.000052871652,0.00010452975,0.0005371105,0.00136773],"genre_scores_gemma":[0.6505452,0.0006539818,0.3429028,0.00020817619,0.0000853762,0.00042718323,0.00058952015,0.00014432435,0.004443439],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99881905,0.00041671723,0.000083626684,0.00021532679,0.00033975113,0.00012560822],"domain_scores_gemma":[0.9954163,0.0032951378,0.00023307672,0.00047260756,0.00046863197,0.00011434813],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00200748,0.00095893384,0.0019956639,0.0008045473,0.0003217821,0.0013360147,0.0021188143,0.0015807645,0.004511326],"category_scores_gemma":[0.012254808,0.0005944119,0.00080977014,0.0011811903,0.0010315286,0.0024432822,0.0019302672,0.002244245,0.0012250439],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011048185,0.00004668709,0.00081860565,0.0001232481,0.00003375923,0.000056121462,0.00007010454,0.87565625,0.0007994117,0.022609858,0.0018412253,0.097834304],"study_design_scores_gemma":[0.0000075827734,0.00001669333,0.00003147378,0.000005709521,0.0000021670464,0.00000945649,0.000005184121,0.9918399,0.00017269146,0.0077042608,0.00020177782,0.0000029873029],"about_ca_topic_score_codex":0.003320387,"about_ca_topic_score_gemma":0.0026739202,"teacher_disagreement_score":0.004511326,"about_ca_system_score_codex":0.0012620552,"about_ca_system_score_gemma":0.0011499467,"threshold_uncertainty_score":0.015091896},"labels":[],"label_agreement":null},{"id":"W2122633037","doi":"10.1109/cdc.1989.70344","title":"Computationally efficient adaptive control algorithms for Markov chains","year":2003,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institut National de la Recherche Scientifique; McGill University","funders":"","keywords":"Markov chain; A priori and a posteriori; Computer science; Computation; Optimal control; Markov decision process; Algorithm; Mathematical optimization; State (computer science); Markov process; Theoretical computer science; Mathematics; Machine learning","score_opus":0.02201706201798436,"score_gpt":0.2565755941705012,"score_spread":0.23455853215251682,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2122633037","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0015614004,0.00025691095,0.99655837,0.00009934732,0.000036531008,0.000043759555,0.000023025803,0.00025357713,0.0011669578],"genre_scores_gemma":[0.26796824,0.0011420534,0.723323,0.0002128847,0.00018985692,0.0010706246,0.00029352185,0.00021215972,0.005587706],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9988398,0.000372119,0.00007445077,0.00022601112,0.00035821265,0.0001295706],"domain_scores_gemma":[0.99636084,0.0026685032,0.0002575638,0.00024086176,0.00037894156,0.00009345527],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017003882,0.0012626355,0.0012869064,0.0010149451,0.0008445724,0.0015970339,0.0018968533,0.0016025557,0.0061470144],"category_scores_gemma":[0.008551912,0.00077166746,0.0008046938,0.0011096246,0.0015339055,0.0020861211,0.0019463144,0.0028828075,0.0009551902],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000063710955,0.000050225823,0.00025333712,0.000116116666,0.000033939214,0.000038079124,0.0000760825,0.78162074,0.0006872373,0.12646838,0.001871674,0.08872044],"study_design_scores_gemma":[0.0000203656,0.000010026169,0.00002664799,0.000010928202,0.0000035207481,0.000009450943,0.0000041214744,0.95777607,0.00017332697,0.041226067,0.00073408236,0.0000054442594],"about_ca_topic_score_codex":0.00564882,"about_ca_topic_score_gemma":0.005301626,"teacher_disagreement_score":0.0061470144,"about_ca_system_score_codex":0.0017947903,"about_ca_system_score_gemma":0.002269489,"threshold_uncertainty_score":0.02056384},"labels":[],"label_agreement":null},{"id":"W2123975568","doi":"10.1109/ictai.2001.974465","title":"Developing collaborative Golog agents by reinforcement learning","year":2001,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Reinforcement learning; Scalability; Human–computer interaction; Plan (archaeology); Knowledge management; Artificial intelligence; Database","score_opus":0.02568505897910089,"score_gpt":0.27940363886921765,"score_spread":0.25371857989011676,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2123975568","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.052448355,0.00014504809,0.94281065,0.00018396995,0.000028955803,0.00015542077,0.000022858854,0.0010022616,0.003202503],"genre_scores_gemma":[0.536665,0.00021778731,0.45883784,0.0001110854,0.000028075057,0.00036405443,0.00009151484,0.00012209585,0.0035624942],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995414,0.00015150724,0.000026791577,0.00009924922,0.00010541333,0.00007555677],"domain_scores_gemma":[0.998831,0.00050767715,0.00019797437,0.00017306604,0.00016056123,0.00012971323],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010321022,0.00085147755,0.0006539732,0.000388988,0.000532942,0.00081725226,0.0014546751,0.0010415278,0.0021544409],"category_scores_gemma":[0.0037667074,0.00049674214,0.00039528625,0.0002551307,0.0011787068,0.0012787903,0.0018002274,0.00085485535,0.0005758752],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012107591,0.00023201642,0.0022610133,0.00012343518,0.00007811197,0.0002774414,0.0004112523,0.8824722,0.008153123,0.022012662,0.0014772968,0.08238041],"study_design_scores_gemma":[0.00003240218,0.00006886074,0.0001137154,0.000010927036,0.00001047479,0.00003496188,0.00003596572,0.9859568,0.0021403425,0.009319239,0.0022673572,0.000009068909],"about_ca_topic_score_codex":0.0027324439,"about_ca_topic_score_gemma":0.0029356692,"teacher_disagreement_score":0.0027324439,"about_ca_system_score_codex":0.00058205536,"about_ca_system_score_gemma":0.0010625401,"threshold_uncertainty_score":0.007207334},"labels":[],"label_agreement":null},{"id":"W2124520701","doi":"10.3233/mgs-2010-0153","title":"Task allocation learning in a multiagent environment: Application to the RoboCupRescue simulation","year":2010,"lang":"en","type":"article","venue":"Multiagent and Grid Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Reinforcement learning; Task (project management); Perception; Human–computer interaction; Artificial intelligence; Multi-agent system; Distributed computing; Machine learning; Systems engineering","score_opus":0.01073146175779477,"score_gpt":0.23421884923192984,"score_spread":0.22348738747413507,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2124520701","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8504168,0.00032876234,0.14029756,0.0007793223,0.000076068616,0.00014893866,0.00016703246,0.0013918768,0.0063937306],"genre_scores_gemma":[0.97714657,0.00006150941,0.021785079,0.00002789359,0.000005608436,0.000053747226,0.000056543075,0.00003485299,0.0008282244],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995895,0.00022866079,0.000015510426,0.000042150656,0.00006749989,0.000056808523],"domain_scores_gemma":[0.9976521,0.0017455384,0.00010392773,0.00014004624,0.00017368322,0.00018471689],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00080912764,0.0005401668,0.00088496256,0.00045144293,0.00049864233,0.00058520935,0.00090200955,0.0010827116,0.0016971494],"category_scores_gemma":[0.0032230755,0.0002535772,0.00035440052,0.00040466528,0.0006836688,0.00047886136,0.0009338226,0.00078743213,0.00013528902],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012422547,0.00012031243,0.00077515404,0.000038647508,0.00002331758,0.00012318495,0.00005709602,0.98955756,0.00063011347,0.0009630577,0.00026876727,0.007318526],"study_design_scores_gemma":[0.000019982377,0.00003316275,0.00014167633,0.0000020053305,0.0000024325993,0.000007822614,0.00001317716,0.99864656,0.00055754837,0.00040157195,0.00017117811,0.0000029856922],"about_ca_topic_score_codex":0.01410582,"about_ca_topic_score_gemma":0.007907333,"teacher_disagreement_score":0.01410582,"about_ca_system_score_codex":0.00078422826,"about_ca_system_score_gemma":0.0006164807,"threshold_uncertainty_score":0.028047442},"labels":[],"label_agreement":null},{"id":"W2126410641","doi":"10.7551/mitpress/7503.003.0060","title":"iLSTD: Eligibility Traces and Convergence Analysis","year":2007,"lang":"en","type":"book-chapter","venue":"The MIT Press eBooks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":42,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Convergence (economics); Computer science; Economics; Macroeconomics","score_opus":0.060792911913390306,"score_gpt":0.29606644664909776,"score_spread":0.23527353473570745,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2126410641","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010146789,0.00048878446,0.98359525,0.0004069667,0.00008244667,0.00007577242,0.00009334756,0.00071761274,0.004392967],"genre_scores_gemma":[0.437069,0.000810205,0.55185497,0.0004565599,0.0001272072,0.00053941034,0.00060570857,0.00089772907,0.007639296],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99774194,0.000779723,0.00013091289,0.00030033774,0.0008383455,0.00020873443],"domain_scores_gemma":[0.9717705,0.021633716,0.00091436843,0.0020558836,0.0029262886,0.0006991511],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00512787,0.0010805118,0.0012625067,0.0016741945,0.0008359517,0.00196575,0.00288157,0.0015848384,0.0089766225],"category_scores_gemma":[0.04484484,0.0007105201,0.001128933,0.0014958803,0.0024132654,0.004630116,0.0032722575,0.0047618276,0.0011836963],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029287604,0.00022630498,0.0026156113,0.0003405255,0.000071455055,0.0001489649,0.00024853187,0.64658654,0.001844139,0.18283714,0.007950055,0.15683787],"study_design_scores_gemma":[0.000012963178,0.00002021317,0.000077278084,0.000019582965,0.0000063227394,0.000023843115,0.0000112268835,0.96909404,0.00059598585,0.029306093,0.00082562084,0.0000067048054],"about_ca_topic_score_codex":0.00465646,"about_ca_topic_score_gemma":0.0034372641,"teacher_disagreement_score":0.0089766225,"about_ca_system_score_codex":0.002044169,"about_ca_system_score_gemma":0.0029993155,"threshold_uncertainty_score":0.030029774},"labels":[],"label_agreement":null},{"id":"W2126848223","doi":"10.1007/978-3-642-15880-3_19","title":"Smarter Sampling in Model-Based Bayesian Reinforcement Learning","year":2010,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Office of Naval Research; Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Reinforcement learning; Markov decision process; Approximate Bayesian computation; Bellman equation; Bayesian probability; Computation; Sample (material); Sampling (signal processing); Artificial intelligence; Posterior probability; Bayesian inference; Machine learning; Markov process; Algorithm; Mathematical optimization; Mathematics; Statistics","score_opus":0.02126356525934251,"score_gpt":0.25524624714224586,"score_spread":0.23398268188290336,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2126848223","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006517508,0.000269849,0.9912019,0.00018557473,0.000043361073,0.000026518002,0.00002083362,0.00016759717,0.001566914],"genre_scores_gemma":[0.6128083,0.00059130025,0.37838686,0.00039106296,0.00021283684,0.00037906968,0.0001919607,0.00028624324,0.006752436],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99731463,0.0015440088,0.00011869771,0.00033689983,0.0005280192,0.00015780205],"domain_scores_gemma":[0.98769116,0.0103512,0.00034211396,0.00085326855,0.00049306545,0.00026909975],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052216756,0.0009284764,0.0025375434,0.00066245557,0.0005498362,0.0013700619,0.0027284848,0.0019691535,0.004748029],"category_scores_gemma":[0.020964045,0.0013190642,0.00095867435,0.00097168115,0.0024874867,0.0034260121,0.0027728807,0.003723845,0.00058555],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020108119,0.000100349156,0.00041879795,0.00011657114,0.00005785148,0.000042928397,0.00009519006,0.81131196,0.00063646195,0.13241461,0.0015335765,0.053070564],"study_design_scores_gemma":[0.000015421514,0.000012790603,0.000025073708,0.000004750031,0.000004394222,0.0000053162626,0.000002131448,0.9549452,0.000100408535,0.04472211,0.00015855281,0.000003952769],"about_ca_topic_score_codex":0.0035926295,"about_ca_topic_score_gemma":0.003777498,"teacher_disagreement_score":0.0052216756,"about_ca_system_score_codex":0.0014621706,"about_ca_system_score_gemma":0.0012482337,"threshold_uncertainty_score":0.02761519},"labels":[],"label_agreement":null},{"id":"W2127186087","doi":"10.7282/t3zw1qcr","title":"Exploring compact reinforcement-learning representations with linear regression","year":2009,"lang":"en","type":"article","venue":"View","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":79,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Linear regression; Artificial intelligence; Function (biology); Regression; Object (grammar); Machine learning; Mathematical optimization; Mathematics; Statistics","score_opus":0.10293541427455874,"score_gpt":0.31457327685415964,"score_spread":0.2116378625796009,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2127186087","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008624167,0.00010153005,0.9901445,0.00009570039,0.000008257055,0.000014597526,0.000014391704,0.00041570308,0.00058109383],"genre_scores_gemma":[0.58817685,0.0002578664,0.40781555,0.00020019217,0.000038072583,0.000245586,0.00019221526,0.00026758824,0.0028060302],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993482,0.00028440746,0.000031366213,0.00014454348,0.00012647468,0.000064950786],"domain_scores_gemma":[0.99790573,0.0015742339,0.0001601806,0.00018186221,0.00013313956,0.00004477707],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014475039,0.0008362883,0.0014118663,0.00044167944,0.00029842526,0.0009757004,0.0013998529,0.0010551754,0.0024275237],"category_scores_gemma":[0.006962114,0.0006412147,0.0005830981,0.0006103072,0.0011983231,0.002630588,0.0018171923,0.001819271,0.0006401791],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008327608,0.000065683016,0.00036451523,0.00007777158,0.000029644812,0.000051920215,0.00010117841,0.8828509,0.0012060207,0.04445457,0.0009837061,0.06973081],"study_design_scores_gemma":[0.000008919291,0.000011244694,0.000012038363,0.0000033585552,0.0000019489878,0.000006223932,0.000005151806,0.9869016,0.00020193742,0.012666495,0.0001787622,0.0000023691223],"about_ca_topic_score_codex":0.0022045192,"about_ca_topic_score_gemma":0.0015143573,"teacher_disagreement_score":0.0024275237,"about_ca_system_score_codex":0.0008265143,"about_ca_system_score_gemma":0.000870304,"threshold_uncertainty_score":0.008120835},"labels":[],"label_agreement":null},{"id":"W2128371572","doi":"10.65109/yget7348","title":"Point-based incremental pruning heuristic for solving finite-horizon DEC-POMDPs","year":2009,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Partially observable Markov decision process; Computer science; Backup; Pruning; Heuristics; Mathematical optimization; Heuristic; Computation; Markov decision process; Bounded function; Markov process; Algorithm; Artificial intelligence; Markov chain; Machine learning; Mathematics; Markov model","score_opus":0.028466300042394005,"score_gpt":0.27550534672508825,"score_spread":0.24703904668269425,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2128371572","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06963231,0.00064694905,0.9253576,0.00018156986,0.00003156782,0.000119589335,0.000104141385,0.0007704362,0.0031558552],"genre_scores_gemma":[0.687546,0.00033450822,0.31031764,0.0000973208,0.000020688038,0.00036160668,0.00029309158,0.00008854477,0.00094061333],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99949825,0.00017986023,0.000027151702,0.00005852844,0.00015268894,0.000083556275],"domain_scores_gemma":[0.99802375,0.0014998177,0.00012622638,0.00012404812,0.00014811657,0.00007811107],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00132042,0.00072130153,0.0014027535,0.00065559946,0.00042226652,0.00057605666,0.0011895916,0.00081597845,0.0013027224],"category_scores_gemma":[0.003997515,0.0005279623,0.00056218833,0.00060824805,0.0006353505,0.0008207948,0.00090080156,0.0010894592,0.00018520457],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007701105,0.000042517822,0.0003702221,0.00008275541,0.000025974618,0.00004758883,0.00003449262,0.9652105,0.0006108971,0.0039103082,0.00039916078,0.029188525],"study_design_scores_gemma":[0.00001877381,0.000030745956,0.0000698994,0.0000074695427,0.0000064721726,0.000010622884,0.000008235332,0.9964276,0.00040495777,0.002808535,0.00020361517,0.000003077589],"about_ca_topic_score_codex":0.0038622408,"about_ca_topic_score_gemma":0.0045556836,"teacher_disagreement_score":0.0038622408,"about_ca_system_score_codex":0.00070643355,"about_ca_system_score_gemma":0.0013348559,"threshold_uncertainty_score":0.007679522},"labels":[],"label_agreement":null},{"id":"W2128619633","doi":"10.1007/978-3-540-30115-8_53","title":"Batch Reinforcement Learning with State Importance","year":2004,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; University of Alberta","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Classifier (UML); Machine learning; Process (computing); State (computer science); Learning classifier system; Q-learning; Quality (philosophy); Function (biology); Bellman equation; Mathematical optimization; Algorithm; Mathematics","score_opus":0.011927894267546134,"score_gpt":0.22735911585792695,"score_spread":0.2154312215903808,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2128619633","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006421003,0.00028860598,0.98754764,0.00012549253,0.00012203844,0.00004824861,0.000037402442,0.00072285696,0.0046865824],"genre_scores_gemma":[0.66342723,0.00040671125,0.31166562,0.0002666588,0.00021816717,0.00034513415,0.00025034684,0.00029100568,0.023129033],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99953187,0.00011447781,0.000025410622,0.000115598916,0.00015707932,0.000055643824],"domain_scores_gemma":[0.99856263,0.00095887453,0.00006873401,0.00017324163,0.00017251637,0.000064051615],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013667659,0.00083223014,0.0011228797,0.00036550098,0.00026223183,0.0005931002,0.0015927341,0.0008964766,0.0072989245],"category_scores_gemma":[0.003479426,0.00052955066,0.00041215424,0.0004411987,0.00076331117,0.0012456984,0.0012991731,0.0018626943,0.0010611219],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004194579,0.00022653058,0.00042052448,0.00017391394,0.000082187325,0.000080958824,0.000050045146,0.61917096,0.0058142594,0.042174395,0.0071413084,0.3242455],"study_design_scores_gemma":[0.000025339747,0.00004351665,0.000060574283,0.000004577035,0.000009102373,0.000014494247,0.0000015569873,0.986206,0.0010498927,0.0119749755,0.00060461415,0.000005250054],"about_ca_topic_score_codex":0.002108066,"about_ca_topic_score_gemma":0.0022401027,"teacher_disagreement_score":0.0072989245,"about_ca_system_score_codex":0.0007020369,"about_ca_system_score_gemma":0.0008109084,"threshold_uncertainty_score":0.024417281},"labels":[],"label_agreement":null},{"id":"W2128812357","doi":"","title":"Error Propagation for Approximate Policy and Value Iteration","year":2010,"lang":"en","type":"preprint","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":135,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Value (mathematics); Computer science; Mathematical optimization; Algorithm; Mathematics; Machine learning","score_opus":0.016255203649768592,"score_gpt":0.2649227238120397,"score_spread":0.2486675201622711,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2128812357","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007718309,0.00026175648,0.98901784,0.00033110206,0.00004373981,0.000031564814,0.000021493044,0.00018385772,0.0023902934],"genre_scores_gemma":[0.6406959,0.0005838335,0.3483793,0.0004550149,0.00015042897,0.00034979486,0.0001540962,0.00030829923,0.008923279],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9957072,0.0017028367,0.00018744866,0.0005658721,0.0014288637,0.0004077978],"domain_scores_gemma":[0.9775557,0.017265176,0.001505102,0.0013446818,0.0019307454,0.00039859602],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007904045,0.0014989745,0.0017640868,0.0012140753,0.00073771225,0.002273334,0.002082408,0.002689722,0.004207565],"category_scores_gemma":[0.043486826,0.00078576704,0.0007874119,0.0012176068,0.0037496754,0.003928559,0.0037822349,0.004313414,0.0007254895],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000216248,0.00005781836,0.0006262692,0.00014313008,0.000051355757,0.0000601471,0.00014139984,0.7781641,0.0012809869,0.17632668,0.0010522123,0.04187969],"study_design_scores_gemma":[0.000007244047,0.000021774535,0.00003587197,0.0000119806145,0.0000039026336,0.0000071962327,0.0000042819884,0.96926516,0.00046842126,0.029933423,0.00023504435,0.00000567531],"about_ca_topic_score_codex":0.00515839,"about_ca_topic_score_gemma":0.0033524702,"teacher_disagreement_score":0.007904045,"about_ca_system_score_codex":0.0032850276,"about_ca_system_score_gemma":0.0025259398,"threshold_uncertainty_score":0.041801035},"labels":[],"label_agreement":null},{"id":"W2128862550","doi":"10.65109/mphg3605","title":"Multi-Agent Patrolling with Reinforcement Learning","year":2004,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":107,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Patrolling; Reinforcement learning; Computer science; Task (project management); Adaptation (eye); Domain (mathematical analysis); Variety (cybernetics); Artificial intelligence; Distributed computing; Machine learning; Engineering","score_opus":0.029312571230412754,"score_gpt":0.2582969557574918,"score_spread":0.22898438452707903,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2128862550","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038099125,0.0002460507,0.9583918,0.00022446907,0.000040794457,0.00008090256,0.000017051312,0.00045917192,0.002440626],"genre_scores_gemma":[0.93272805,0.00010472889,0.06497728,0.000077204786,0.000034225555,0.00014640475,0.000030508321,0.00003227414,0.0018693061],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993843,0.000275309,0.000030096397,0.000122066995,0.000115975585,0.00007231493],"domain_scores_gemma":[0.9979869,0.0012021313,0.00029140577,0.00018961015,0.00018909079,0.000140979],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014578557,0.0009830096,0.0011366095,0.0003544933,0.00043370837,0.000673196,0.0016149638,0.001241876,0.0017387948],"category_scores_gemma":[0.0037451056,0.00040342117,0.0004779001,0.00033157374,0.0013465809,0.0009280466,0.001248494,0.0013363557,0.00033227625],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000045438206,0.000056056553,0.00035791527,0.000025773437,0.000026078365,0.00004373744,0.000024676823,0.9832812,0.0005131889,0.0027873728,0.00020215561,0.012636395],"study_design_scores_gemma":[0.000013619606,0.00001933021,0.000030379264,0.0000014399918,0.000002330251,0.0000059842287,0.0000021139679,0.99821955,0.0001322755,0.0014639965,0.000107135595,0.000001722426],"about_ca_topic_score_codex":0.0040296004,"about_ca_topic_score_gemma":0.0024840366,"teacher_disagreement_score":0.0040296004,"about_ca_system_score_codex":0.00070628815,"about_ca_system_score_gemma":0.0008589567,"threshold_uncertainty_score":0.008012295},"labels":[],"label_agreement":null},{"id":"W2129479396","doi":"10.1109/tsmcb.2003.821869","title":"Modular Fuzzy-Reinforcement Learning Approach With Internal Model Capabilities for Multiagent Systems","year":2004,"lang":"en","type":"article","venue":"IEEE Transactions on Systems Man and Cybernetics Part B (Cybernetics)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Modular design; Computer science; Reinforcement learning; Robustness (evolution); Fuzzy logic; Artificial intelligence; Multi-agent system; Architecture; State space; Domain (mathematical analysis); Internal model; Control (management); Mathematics","score_opus":0.022985664478799343,"score_gpt":0.22860031637009612,"score_spread":0.20561465189129677,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2129479396","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0076460233,0.00013414242,0.99059314,0.00008862891,0.000017131742,0.0000205412,0.00001008486,0.00014116526,0.0013492313],"genre_scores_gemma":[0.7936647,0.0002837783,0.20356183,0.00007015496,0.00006330518,0.00017770435,0.000045111457,0.000033692646,0.0020998635],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995572,0.00014994614,0.000023320512,0.00007905601,0.00014804085,0.000042398642],"domain_scores_gemma":[0.99941814,0.00025652503,0.00008404465,0.0000770668,0.00012411995,0.000040108866],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010485346,0.0007266714,0.0007997185,0.0004124712,0.0003487507,0.000691304,0.0011207589,0.00078367244,0.0018001195],"category_scores_gemma":[0.0018573542,0.00027863227,0.0008056152,0.00032249247,0.0007384602,0.0009520357,0.0009751084,0.001121037,0.0003116056],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003470994,0.00004782459,0.00031870577,0.00006832007,0.000059761067,0.000101602054,0.00009785381,0.9194805,0.00217043,0.04229224,0.00039997333,0.034928028],"study_design_scores_gemma":[0.000008460437,0.000025833051,0.0000411404,0.0000036936212,0.000007017582,0.000012472109,0.0000036177269,0.9877771,0.00030598865,0.011482173,0.00032825474,0.0000042422034],"about_ca_topic_score_codex":0.0024868276,"about_ca_topic_score_gemma":0.0019379146,"teacher_disagreement_score":0.0024868276,"about_ca_system_score_codex":0.00083518296,"about_ca_system_score_gemma":0.00061549037,"threshold_uncertainty_score":0.006059706},"labels":[],"label_agreement":null},{"id":"W2129578296","doi":"10.1109/cccrv.2004.1301489","title":"A reinforcement learning framework for parameter control in computer vision applications","year":2004,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Entropy (arrow of time); Selection (genetic algorithm); Fuzzy logic; Principle of maximum entropy; Function approximation; Machine learning; Artificial neural network","score_opus":0.013490645539640269,"score_gpt":0.28086394825622957,"score_spread":0.2673733027165893,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2129578296","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013011185,0.0003011906,0.99718094,0.000106785556,0.000025486774,0.000022906215,0.000006221014,0.00011136775,0.000944015],"genre_scores_gemma":[0.4851757,0.0011220457,0.5072646,0.0002276754,0.00021103634,0.0005349446,0.00006229578,0.000094538664,0.0053070444],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993864,0.00022152252,0.000031714346,0.00010505939,0.00019957907,0.000055746415],"domain_scores_gemma":[0.999411,0.0002862829,0.00007095654,0.000047974758,0.00012919308,0.000054574353],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013628133,0.0009885399,0.0012221716,0.00039045102,0.0004584923,0.00088149443,0.0019069922,0.0012864775,0.0022601334],"category_scores_gemma":[0.001864374,0.00040979145,0.00078442367,0.00047973898,0.0014849225,0.0009850913,0.0009225691,0.0019855334,0.0004780454],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003864677,0.000059893377,0.00013807246,0.000085814165,0.00004323876,0.00009791046,0.000060817416,0.87136054,0.0030377624,0.075346075,0.0010503437,0.048680864],"study_design_scores_gemma":[0.000014192846,0.000040908173,0.000037695183,0.000007982867,0.0000065660015,0.000018459608,0.000003861632,0.9771488,0.00048267274,0.020684723,0.0015464319,0.000007880186],"about_ca_topic_score_codex":0.004432071,"about_ca_topic_score_gemma":0.003003182,"teacher_disagreement_score":0.004432071,"about_ca_system_score_codex":0.0011550995,"about_ca_system_score_gemma":0.0011446881,"threshold_uncertainty_score":0.008812547},"labels":[],"label_agreement":null},{"id":"W2132622533","doi":"10.5555/2031678.2031726","title":"Horde: a scalable real-time architecture for learning knowledge from unsupervised sensorimotor interaction","year":2011,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":305,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; University of Alberta","funders":"","keywords":"Computer science; Artificial intelligence; Reinforcement learning; Function (biology); Scalability; Temporal difference learning; Modular design; Architecture; Robot; Robot learning; Robotics; Machine learning; Unsupervised learning; Mobile robot","score_opus":0.032686465502922035,"score_gpt":0.25832884638991926,"score_spread":0.22564238088699723,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2132622533","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012877752,0.00015001232,0.98085314,0.00013133476,0.000039473915,0.000082079416,0.000062866486,0.004196887,0.0016064325],"genre_scores_gemma":[0.42927045,0.0002777332,0.5621316,0.0002621163,0.000033747947,0.00041093348,0.00047026717,0.000290218,0.0068529914],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99966514,0.00004551597,0.000019826966,0.0001116454,0.00011293107,0.000044901557],"domain_scores_gemma":[0.999398,0.0002003988,0.000051478255,0.00017245334,0.00011883919,0.00005885834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008872101,0.00067106483,0.0007572792,0.00029826115,0.00036603765,0.00069498277,0.003404717,0.00082713796,0.003534104],"category_scores_gemma":[0.0017012602,0.00059130584,0.0005085054,0.00026766115,0.00090962654,0.001797557,0.0020418132,0.001985354,0.0009462184],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003914035,0.00034805806,0.0014332818,0.0002021473,0.00015669696,0.00018045722,0.00017066555,0.5644562,0.021533852,0.015540449,0.006380974,0.38920575],"study_design_scores_gemma":[0.00002782707,0.000046805257,0.00012534877,0.0000047813346,0.000011273839,0.000024513058,0.000009326564,0.9881544,0.0041513178,0.005598011,0.0018374689,0.0000088583265],"about_ca_topic_score_codex":0.004233822,"about_ca_topic_score_gemma":0.007976656,"teacher_disagreement_score":0.004233822,"about_ca_system_score_codex":0.0007552839,"about_ca_system_score_gemma":0.0012916452,"threshold_uncertainty_score":0.01182276},"labels":[],"label_agreement":null},{"id":"W2133342902","doi":"10.1109/ijcnn.2006.246687","title":"Learning to Coordinate Behaviors in Soft Behavior-Based Systems Using Reinforcement Learning","year":2006,"lang":"en","type":"article","venue":"The 2006 IEEE International Joint Conference on Neural Network Proceedings","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Architecture; Task (project management); Mobile robot; Robot; Behavior-based robotics; Robotics; Mechanism (biology); Reinforcement; Engineering","score_opus":0.0449721869003996,"score_gpt":0.28035691451561023,"score_spread":0.23538472761521062,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2133342902","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.083433464,0.000102256396,0.91356623,0.0001986215,0.000030159155,0.00009378172,0.000013168926,0.00057211524,0.0019902831],"genre_scores_gemma":[0.93865085,0.00006791815,0.059950523,0.00008623233,0.000016900329,0.00013894044,0.000020738275,0.000026783984,0.0010410666],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995216,0.0001459341,0.000038030652,0.00010060147,0.00013405198,0.000059755228],"domain_scores_gemma":[0.9986131,0.0005636169,0.0003120179,0.0001671233,0.00022019944,0.00012388226],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010008047,0.0007351299,0.00053007534,0.00033491306,0.0003081594,0.0005181049,0.0007525151,0.0004863166,0.0010318363],"category_scores_gemma":[0.0028688535,0.00029603727,0.00035844295,0.00019633965,0.0012432466,0.0008271726,0.00094281905,0.0009621471,0.0001882126],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001237425,0.00017683525,0.0018116048,0.000108273445,0.00008492501,0.00015442655,0.00025172648,0.8839956,0.02480746,0.019088374,0.0005216804,0.068875335],"study_design_scores_gemma":[0.000022873483,0.000065475935,0.00017235547,0.0000044215753,0.000007833988,0.000015166796,0.000008544636,0.99058163,0.0015348843,0.007366616,0.00021335043,0.000006849184],"about_ca_topic_score_codex":0.0018316907,"about_ca_topic_score_gemma":0.0020318977,"teacher_disagreement_score":0.0018316907,"about_ca_system_score_codex":0.00055554317,"about_ca_system_score_gemma":0.0008101429,"threshold_uncertainty_score":0.005292833},"labels":[],"label_agreement":null},{"id":"W2133552775","doi":"","title":"Learning from Limited Demonstrations","year":2013,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":72,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Variety (cybernetics); Task (project management); Key (lock); Path (computing); Artificial intelligence; Machine learning; Mathematical optimization; Temporal difference learning; Mathematics","score_opus":0.01983794697850236,"score_gpt":0.22507605906335879,"score_spread":0.20523811208485643,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2133552775","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02522689,0.00020136627,0.9721025,0.00020091071,0.000021470978,0.000041159947,0.00006336283,0.0006312011,0.0015111613],"genre_scores_gemma":[0.7503097,0.00017166798,0.24610099,0.00019860332,0.000039520666,0.00023231107,0.000287163,0.00011943121,0.0025406538],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988212,0.00046206824,0.00006321898,0.00029034892,0.00027434135,0.000088833585],"domain_scores_gemma":[0.99293023,0.0047662463,0.00057143974,0.0008310479,0.00061792845,0.00028319738],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020624795,0.0007929308,0.0013473808,0.00052962435,0.00038092257,0.0009137292,0.0019892135,0.0013442072,0.0028463276],"category_scores_gemma":[0.015189405,0.00084446376,0.0004761583,0.00043025264,0.0012344897,0.0024342593,0.0021875645,0.0018662848,0.00063183246],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017644351,0.00007799525,0.0012285163,0.0001357615,0.000047778227,0.00011085244,0.00011597797,0.8797263,0.0023704723,0.015307438,0.0014129933,0.09928946],"study_design_scores_gemma":[0.000014007825,0.000029422697,0.00007562981,0.000008254283,0.0000030873723,0.000015675576,0.0000054496636,0.9926934,0.0005388372,0.00632985,0.0002811763,0.000005074377],"about_ca_topic_score_codex":0.0024910837,"about_ca_topic_score_gemma":0.002838331,"teacher_disagreement_score":0.0028463276,"about_ca_system_score_codex":0.0007423583,"about_ca_system_score_gemma":0.0014440536,"threshold_uncertainty_score":0.01090759},"labels":[],"label_agreement":null},{"id":"W2134042548","doi":"","title":"Convergent Temporal-Difference Learning with Arbitrary Smooth Function Approximation","year":2009,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":168,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; McGill University","funders":"","keywords":"Temporal difference learning; Function approximation; Stochastic approximation; Markov decision process; Mathematics; Function (biology); Stochastic gradient descent; Bellman equation; Mathematical optimization; Convergence (economics); Nonlinear system; Approximation algorithm; Artificial neural network; Gradient descent; Approximation error; Markov process; Rate of convergence; Algorithm; Applied mathematics; Reinforcement learning; Computer science; Artificial intelligence; Key (lock)","score_opus":0.014547591731831377,"score_gpt":0.22033568944607174,"score_spread":0.20578809771424036,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2134042548","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0031189627,0.00018422416,0.99502325,0.00011417312,0.000041142124,0.000021424565,0.000012786874,0.000084561274,0.0013995488],"genre_scores_gemma":[0.36308494,0.00044978608,0.62909395,0.00024171559,0.00008954435,0.00023373024,0.0000930766,0.00013022129,0.006583024],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993286,0.00017568188,0.000042901785,0.00013583571,0.0002541276,0.00006286525],"domain_scores_gemma":[0.9976478,0.0013774668,0.00018216581,0.00025522197,0.0004383454,0.00009886328],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002428547,0.0008440595,0.0010280835,0.00051005225,0.00040181432,0.0010917281,0.0017996329,0.0014512967,0.0028723846],"category_scores_gemma":[0.008947497,0.00047191174,0.00082487863,0.0005218561,0.0017629805,0.0018871228,0.0020687948,0.0023006564,0.00055823964],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000118207696,0.00006276508,0.00059133733,0.00015366606,0.000046921676,0.00009520556,0.00012632922,0.70481735,0.0026210784,0.22415246,0.0010968762,0.06611783],"study_design_scores_gemma":[0.0000064694386,0.000017281607,0.000021528036,0.0000058626706,0.0000027019448,0.000012807204,0.00000210343,0.9857782,0.00048005427,0.013133815,0.0005346536,0.0000044861663],"about_ca_topic_score_codex":0.0031792736,"about_ca_topic_score_gemma":0.0019369519,"teacher_disagreement_score":0.0031792736,"about_ca_system_score_codex":0.0013477986,"about_ca_system_score_gemma":0.0013070058,"threshold_uncertainty_score":0.012843549},"labels":[],"label_agreement":null},{"id":"W2135097260","doi":"10.1007/3-540-44886-1_26","title":"Model-Based Least-Squares Policy Evaluation","year":2003,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Estimator; Computation; Least-squares function approximation; Basis (linear algebra); Algorithm; Set (abstract data type); Markov decision process; Mathematical optimization; Markov process; Mathematics; Statistics","score_opus":0.03695932473739669,"score_gpt":0.29243450260143466,"score_spread":0.25547517786403795,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2135097260","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011803143,0.00025903856,0.98315465,0.00025917694,0.00010116523,0.00009034839,0.000070943875,0.0017283027,0.0025332707],"genre_scores_gemma":[0.60628825,0.0001863423,0.3857311,0.0003383873,0.0001038313,0.00036958006,0.00046222494,0.00060726056,0.0059129368],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.997829,0.0010717843,0.00012418382,0.0003072908,0.00044785993,0.00021982203],"domain_scores_gemma":[0.99094373,0.0064854864,0.0003165701,0.00043543798,0.001583503,0.00023522892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00447492,0.0016914787,0.003368322,0.0012860456,0.0009250512,0.0020428377,0.0022376315,0.003436327,0.008252558],"category_scores_gemma":[0.016656836,0.0013967282,0.0010184898,0.00085397693,0.0010662291,0.0018975448,0.00187127,0.0023934552,0.0017963364],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002642635,0.00008439445,0.00036023525,0.00008455896,0.000055614586,0.00003218612,0.000021681562,0.9292284,0.0005290694,0.002878066,0.0017487945,0.064712696],"study_design_scores_gemma":[0.000012168946,0.0000130901935,0.0000202146,0.000003425568,0.000003933445,0.00000366173,0.0000023430266,0.99886346,0.00021744914,0.00079593837,0.00006168731,0.0000026382907],"about_ca_topic_score_codex":0.01230198,"about_ca_topic_score_gemma":0.00945727,"teacher_disagreement_score":0.01230198,"about_ca_system_score_codex":0.0017067612,"about_ca_system_score_gemma":0.0036499908,"threshold_uncertainty_score":0.02760756},"labels":[],"label_agreement":null},{"id":"W2136602922","doi":"","title":"Incremental Natural Actor-Critic Algorithms","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":157,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Temporal difference learning; Computer science; Convergence (economics); Bellman equation; Mathematical proof; Function approximation; Algorithm; Variance (accounting); Stochastic gradient descent; Function (biology); Rate of convergence; Gradient descent; Artificial intelligence; Mathematical optimization; Mathematics; Artificial neural network","score_opus":0.013354005743277607,"score_gpt":0.2696630519567899,"score_spread":0.2563090462135123,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2136602922","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032450985,0.00022108523,0.9911856,0.00013996042,0.000088487286,0.000069977905,0.00003736457,0.00051847496,0.004493989],"genre_scores_gemma":[0.29504004,0.00039509113,0.6914247,0.0002830343,0.00011782151,0.0005111451,0.00020107169,0.00019162933,0.011835431],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989297,0.0003148904,0.000062364925,0.00020427817,0.00038482153,0.00010388333],"domain_scores_gemma":[0.99830294,0.00077144435,0.00015190322,0.00026218928,0.00038858122,0.00012299234],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016885583,0.001099574,0.0011655771,0.0006874039,0.00056083774,0.0010521745,0.0027516244,0.0012267067,0.0053866087],"category_scores_gemma":[0.0056873765,0.00059867505,0.00071317796,0.00045683837,0.0011259039,0.0018809867,0.0020066146,0.0017998943,0.0011007776],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001894272,0.00014957739,0.0008266749,0.00025899432,0.000093972274,0.00013275792,0.00018085839,0.5745934,0.0024804983,0.20070164,0.008456803,0.21193527],"study_design_scores_gemma":[0.000022613269,0.000029201963,0.00005919015,0.000007855203,0.000009709608,0.00003343178,0.000005254221,0.97420233,0.00054011715,0.021620398,0.003459274,0.000010583252],"about_ca_topic_score_codex":0.0026185438,"about_ca_topic_score_gemma":0.0036216632,"teacher_disagreement_score":0.0053866087,"about_ca_system_score_codex":0.0011281269,"about_ca_system_score_gemma":0.0014981084,"threshold_uncertainty_score":0.018020034},"labels":[],"label_agreement":null},{"id":"W2136617285","doi":"10.1109/icsmc.2007.4414024","title":"Aggregation of tiling-based reinforcement learning algorithms","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Aggregate (composite); Architecture; Artificial intelligence; Function (biology); Stability (learning theory); Instance-based learning; Competitive learning; Algorithm; Machine learning; Unsupervised learning","score_opus":0.01789302460037047,"score_gpt":0.2669956302511332,"score_spread":0.24910260565076275,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2136617285","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04560184,0.0002917563,0.9494561,0.00012743943,0.00008787457,0.00009256371,0.000024885401,0.00092568074,0.003391864],"genre_scores_gemma":[0.84704036,0.00018291375,0.14984722,0.000119175995,0.000057204503,0.00016120145,0.000082690734,0.00009014529,0.0024190526],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99886715,0.00034133322,0.000091035865,0.00026785265,0.00030463914,0.00012799934],"domain_scores_gemma":[0.9976299,0.0010753066,0.00026508662,0.00036260666,0.00052057346,0.000146541],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019328357,0.0007870218,0.0015062569,0.0006926321,0.00043684692,0.00085522037,0.0011377231,0.00066583493,0.002668949],"category_scores_gemma":[0.0053750225,0.00039589228,0.0006100188,0.00049962173,0.0009135202,0.0013417093,0.0018891349,0.0010100175,0.00043639762],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012093924,0.00007830176,0.0011891419,0.000061515406,0.0000715298,0.00007853495,0.00012727862,0.8721768,0.0032083644,0.007677966,0.0007599849,0.11444968],"study_design_scores_gemma":[0.000011444717,0.000049673872,0.00007837897,0.0000044214676,0.000007219056,0.000013067199,0.000006291951,0.9951565,0.00069045,0.0035058511,0.00047242793,0.0000043115388],"about_ca_topic_score_codex":0.0021656642,"about_ca_topic_score_gemma":0.0016825629,"teacher_disagreement_score":0.002668949,"about_ca_system_score_codex":0.0007462999,"about_ca_system_score_gemma":0.00062183803,"threshold_uncertainty_score":0.010221899},"labels":[],"label_agreement":null},{"id":"W2136895343","doi":"","title":"Brain Inspired Reinforcement Learning","year":2004,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Neurophysiology; Machine learning; Adaptation (eye); Task (project management); Convergence (economics); Learning classifier system; Feature (linguistics); Neuroscience; Psychology; Engineering","score_opus":0.016273083672864667,"score_gpt":0.24571917080806616,"score_spread":0.22944608713520148,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2136895343","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019062832,0.0014773479,0.96610105,0.00053172046,0.00017312489,0.00009568546,0.000041632316,0.0005069823,0.012009626],"genre_scores_gemma":[0.8317022,0.0012149558,0.16038588,0.0003558781,0.000098100645,0.0003028075,0.00007633772,0.000054798995,0.005809243],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9997501,0.00009220128,0.000012697697,0.000042197848,0.00007602207,0.000026753327],"domain_scores_gemma":[0.99922645,0.0004981815,0.000074872565,0.00004683297,0.00011239939,0.000041161205],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007128905,0.00060618087,0.00064521894,0.00026777165,0.00024298416,0.0006078939,0.0008264537,0.000817134,0.0025122175],"category_scores_gemma":[0.0026459869,0.00016328556,0.0002977256,0.00023388036,0.00082254544,0.0005059403,0.00062658585,0.0009672274,0.00034951922],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000107607375,0.00011935727,0.0008250391,0.0002030744,0.00009438228,0.00011210675,0.000065768225,0.8042577,0.0036801302,0.060852766,0.002632536,0.12704952],"study_design_scores_gemma":[0.00004371271,0.00009047332,0.0001324361,0.000017485225,0.000014060495,0.000047820755,0.000007061275,0.9655492,0.00094948633,0.029787648,0.0033503922,0.00001023851],"about_ca_topic_score_codex":0.0011148464,"about_ca_topic_score_gemma":0.0011758019,"teacher_disagreement_score":0.0025122175,"about_ca_system_score_codex":0.0006012594,"about_ca_system_score_gemma":0.0005663265,"threshold_uncertainty_score":0.008404195},"labels":[],"label_agreement":null},{"id":"W2136937392","doi":"10.1287/moor.1050.0148","title":"On the Empirical State-Action Frequencies in Markov Decision Processes Under General Policies","year":2005,"lang":"en","type":"article","venue":"Mathematics of Operations Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"National Science Foundation","keywords":"Polytope; Mathematics; Markov decision process; Markov chain; State (computer science); Element (criminal law); Action (physics); Limit (mathematics); Empirical research; Finite state; Markov process; Mathematical optimization; Mathematical economics; Combinatorics; Statistics; Algorithm; Mathematical analysis; Law","score_opus":0.31700725284698666,"score_gpt":0.46017316569362243,"score_spread":0.14316591284663577,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2136937392","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23369938,0.0006927629,0.75780016,0.00081263686,0.000046577297,0.00007434916,0.00019275199,0.00018255346,0.0064988662],"genre_scores_gemma":[0.9342227,0.00086784514,0.061394587,0.0001173793,0.00012507336,0.00026908904,0.00020174317,0.00011055849,0.0026910393],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99622786,0.0016026044,0.00019205408,0.0006974065,0.0008329007,0.00044726665],"domain_scores_gemma":[0.94660753,0.04479291,0.0038843572,0.0018771044,0.0016045807,0.0012335925],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0065195123,0.00081151206,0.0013855161,0.0015743544,0.0010192916,0.002532159,0.0016870825,0.0014355923,0.0036270476],"category_scores_gemma":[0.03610321,0.0008273764,0.0007925252,0.0011884595,0.0047935518,0.005824296,0.0027141825,0.0026372494,0.00035135637],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016358474,0.000072202405,0.0015725814,0.00011175774,0.00004387851,0.0001289329,0.00027694486,0.3607748,0.0020264878,0.62433213,0.0003400574,0.010156719],"study_design_scores_gemma":[0.00002902365,0.00008151016,0.00083608116,0.000046275876,0.00001346819,0.000057994654,0.000072947085,0.6539472,0.0006073636,0.34375054,0.00052253634,0.000035017652],"about_ca_topic_score_codex":0.0018703465,"about_ca_topic_score_gemma":0.0009423223,"teacher_disagreement_score":0.0065195123,"about_ca_system_score_codex":0.0024830773,"about_ca_system_score_gemma":0.0013100667,"threshold_uncertainty_score":0.034478903},"labels":[],"label_agreement":null},{"id":"W2137312721","doi":"","title":"On-line Reinforcement Learning Using Incremental Kernel-Based Stochastic Factorization","year":2012,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Markov decision process; Computer science; Kernel (algebra); Factorization; Q-learning; Markov process; Algorithm; State space; Mathematical optimization; Artificial intelligence; Mathematics","score_opus":0.045062490513218506,"score_gpt":0.28652545567702453,"score_spread":0.24146296516380603,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2137312721","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02028822,0.00007556794,0.9781309,0.00007234803,0.000017239085,0.000052583036,0.000016642269,0.00046167264,0.0008847582],"genre_scores_gemma":[0.82668734,0.00007661138,0.17171752,0.000096359065,0.000020523348,0.00015909999,0.00008656009,0.00006476316,0.0010912401],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99945706,0.00016167785,0.000026934156,0.00010031734,0.0001748488,0.000079109406],"domain_scores_gemma":[0.9975794,0.0015508892,0.0002374814,0.0002054419,0.00031342017,0.000113380614],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012685214,0.0008984014,0.0012686749,0.00035820674,0.00036851602,0.00063521543,0.0014810975,0.000967887,0.0018171655],"category_scores_gemma":[0.005609119,0.00044444203,0.0005417665,0.0002868109,0.0009970631,0.0011313392,0.0012328962,0.0016048547,0.000319654],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009474227,0.00009646386,0.00051997753,0.00004771975,0.000026678339,0.000058315105,0.00005583209,0.9461254,0.0014594065,0.006452494,0.0005957516,0.04446721],"study_design_scores_gemma":[0.0000073034216,0.000016758018,0.000022473305,0.000001538287,0.000001866187,0.000005927744,0.0000015673739,0.99840003,0.00021520114,0.0012561481,0.00006912074,0.0000020267237],"about_ca_topic_score_codex":0.007127197,"about_ca_topic_score_gemma":0.005546447,"teacher_disagreement_score":0.007127197,"about_ca_system_score_codex":0.00089283206,"about_ca_system_score_gemma":0.0014996557,"threshold_uncertainty_score":0.0141714215},"labels":[],"label_agreement":null},{"id":"W2138781282","doi":"10.1613/jair.3384","title":"Policy Invariance under Reward Transformations for General-Sum Stochastic Games","year":2011,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Royal Military College of Canada; Carleton University","funders":"","keywords":"Convergence (economics); Nash equilibrium; Mathematical economics; Property (philosophy); Markov chain; Computer science; Markov decision process; Mathematical optimization; Mathematics; Markov process; Economics; Machine learning","score_opus":0.28393678730286626,"score_gpt":0.42748179026709787,"score_spread":0.14354500296423162,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2138781282","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.057533022,0.00006830904,0.93515867,0.00021212728,0.00003600059,0.000049507096,0.000025875312,0.0001882497,0.0067282435],"genre_scores_gemma":[0.95945543,0.00013630744,0.03681026,0.00009725608,0.000030608382,0.000120856406,0.000039668448,0.00009546015,0.003214194],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99906105,0.00037081967,0.000045854857,0.00014818189,0.0002320433,0.00014209055],"domain_scores_gemma":[0.99685645,0.00201453,0.0003708869,0.0003008242,0.00025586266,0.00020146501],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018298083,0.0005914665,0.0006600526,0.0004161766,0.0004549478,0.0010996074,0.0007873988,0.0007123808,0.0024574834],"category_scores_gemma":[0.010950774,0.00029939105,0.00078481436,0.00028421078,0.0018226712,0.0015855386,0.0018819843,0.0016288063,0.00033370004],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013498898,0.00009390394,0.00053533405,0.00006977589,0.000045498182,0.00019227719,0.0002511871,0.5403879,0.00791953,0.42229164,0.0006125887,0.027465349],"study_design_scores_gemma":[0.000015957434,0.000058389527,0.00011021314,0.0000061738683,0.000005541545,0.000028623845,0.000016739014,0.8426661,0.0011682233,0.1555754,0.0003383801,0.000010236433],"about_ca_topic_score_codex":0.0012823879,"about_ca_topic_score_gemma":0.00060613925,"teacher_disagreement_score":0.0024574834,"about_ca_system_score_codex":0.00081883895,"about_ca_system_score_gemma":0.0009253809,"threshold_uncertainty_score":0.0096770525},"labels":[],"label_agreement":null},{"id":"W2139555052","doi":"10.1287/opre.2014.1332","title":"Myopic Bounds for Optimal Policy of POMDPs: An Extension of Lovejoy’s Structural Results","year":2015,"lang":"en","type":"article","venue":"Operations Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Extension (predicate logic); Bounded function; Markov decision process; Mathematical optimization; Computer science; Relaxation (psychology); Upper and lower bounds; Mathematical economics; Mathematics; Markov process","score_opus":0.22538392064775611,"score_gpt":0.47345286381946544,"score_spread":0.24806894317170933,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2139555052","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01158186,0.00050179905,0.97432977,0.0010659472,0.000055623732,0.00007056233,0.00018412103,0.00017256828,0.012037746],"genre_scores_gemma":[0.75923806,0.0017348311,0.23054934,0.0007647976,0.00035219814,0.00077335146,0.0004963414,0.00040541933,0.0056856107],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9954919,0.0014040391,0.00024581928,0.0007585945,0.0016403801,0.00045927454],"domain_scores_gemma":[0.971943,0.021790957,0.0018618203,0.0018670345,0.0018235795,0.0007135637],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006305123,0.0014741599,0.0017480479,0.0016608561,0.0010690856,0.0022668152,0.002684369,0.0016304129,0.007790913],"category_scores_gemma":[0.03567569,0.0015952757,0.0019550705,0.0011693003,0.0033319928,0.0075087436,0.0042382292,0.005633146,0.0007649835],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000110715664,0.00008255529,0.00043452997,0.00026911285,0.000042875305,0.00009601684,0.00034789857,0.2843559,0.0018170221,0.69266534,0.001779272,0.017998772],"study_design_scores_gemma":[0.00003487602,0.00008162056,0.00023452613,0.000110982575,0.000016157504,0.000045689896,0.000052911968,0.52606267,0.0012171767,0.4692478,0.0028666873,0.000028942975],"about_ca_topic_score_codex":0.0017763883,"about_ca_topic_score_gemma":0.001648782,"teacher_disagreement_score":0.007790913,"about_ca_system_score_codex":0.0024528997,"about_ca_system_score_gemma":0.0030492218,"threshold_uncertainty_score":0.033345103},"labels":[],"label_agreement":null},{"id":"W2140035747","doi":"","title":"Action selection in Bayesian reinforcement learning","year":2006,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Action selection; Machine learning; Computer science; Artificial intelligence; Bayesian probability; Markov decision process; Selection (genetic algorithm); Variable-order Bayesian network; Action (physics); Perspective (graphical); Bayesian inference; Markov process; Mathematics","score_opus":0.014056294869686356,"score_gpt":0.24991752923176236,"score_spread":0.235861234362076,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2140035747","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013918457,0.00034888816,0.981222,0.0005163572,0.000047163478,0.000061219514,0.00004319202,0.0001917352,0.0036509994],"genre_scores_gemma":[0.79592896,0.00058598124,0.1958569,0.00045527046,0.00014963398,0.00046461265,0.00015237542,0.000097691714,0.0063084983],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99768806,0.0012464636,0.00008431232,0.00031541486,0.00048762385,0.00017814976],"domain_scores_gemma":[0.9931463,0.005483787,0.00038440464,0.0002548603,0.0004906025,0.00024001578],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003499861,0.0010852561,0.0016637739,0.00065933465,0.0004818816,0.0011002136,0.0018426295,0.0016016687,0.0038031172],"category_scores_gemma":[0.013708968,0.0007920624,0.0006592702,0.00073227787,0.00204245,0.0024571002,0.0013349075,0.0024408347,0.00055715075],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013229718,0.000107921296,0.00097091385,0.00011885922,0.00007441759,0.000065033404,0.00013762426,0.8443677,0.00061279285,0.10061463,0.0013300397,0.05146778],"study_design_scores_gemma":[0.000030443189,0.00002927355,0.00007668073,0.000010356094,0.000008077681,0.0000086401005,0.0000062306485,0.9555749,0.00015744535,0.04368626,0.00040406728,0.0000076214374],"about_ca_topic_score_codex":0.005285316,"about_ca_topic_score_gemma":0.005499345,"teacher_disagreement_score":0.005285316,"about_ca_system_score_codex":0.0018297499,"about_ca_system_score_gemma":0.0014147459,"threshold_uncertainty_score":0.018509269},"labels":[],"label_agreement":null},{"id":"W2141050197","doi":"10.4304/jcp.2.1.12-19","title":"Noisy K Best-Paths for Approximate Dynamic Programming with Application to Portfolio Optimization","year":2007,"lang":"en","type":"article","venue":"Journal of Computers","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Dynamic programming; Computer science; Sharpe ratio; Reinforcement learning; Mathematical optimization; Portfolio; Markov decision process; Kernel (algebra); Controller (irrigation); Artificial intelligence; Markov process; Mathematics; Algorithm; Finance","score_opus":0.007554112725446023,"score_gpt":0.26092508528742103,"score_spread":0.253370972561975,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2141050197","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006583329,0.000092936985,0.9924873,0.00012417677,0.0000129442315,0.000016932025,0.000011635768,0.00017081425,0.0004999828],"genre_scores_gemma":[0.50016,0.00017241485,0.4971606,0.00012170825,0.000034210036,0.00022706024,0.00007843102,0.00012556666,0.001920046],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991547,0.00043032892,0.00004329907,0.0001399672,0.0001879551,0.000043758777],"domain_scores_gemma":[0.9976115,0.0017774298,0.00016422594,0.00016174217,0.000216577,0.00006854233],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019507117,0.0008842945,0.0012730543,0.00045026644,0.0004555154,0.0008147105,0.00096146745,0.0014266995,0.002285657],"category_scores_gemma":[0.0072349347,0.00069598056,0.00055158156,0.00068842666,0.0013309891,0.0012997375,0.0014274231,0.0017246389,0.0003067717],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000058743124,0.00003795115,0.0002163721,0.00003573561,0.000022375809,0.000028074734,0.000031137515,0.9559172,0.000730832,0.016054893,0.00034411452,0.026522698],"study_design_scores_gemma":[0.000003463382,0.000007925673,0.000013667433,0.00000159269,9.4158526e-7,0.000002775082,9.341144e-7,0.9953969,0.0001493464,0.0043464643,0.00007444484,0.0000014961267],"about_ca_topic_score_codex":0.0027638006,"about_ca_topic_score_gemma":0.0020720924,"teacher_disagreement_score":0.0027638006,"about_ca_system_score_codex":0.0011256459,"about_ca_system_score_gemma":0.0011737574,"threshold_uncertainty_score":0.010316491},"labels":[],"label_agreement":null},{"id":"W2141650360","doi":"10.1109/icsmc.2004.1399805","title":"Reinforcement learning and aggregation","year":2005,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Robustness (evolution); Artificial intelligence; Q-learning; Machine learning; Computation; Learning classifier system; Algorithm","score_opus":0.011181410940896828,"score_gpt":0.24484486953103568,"score_spread":0.23366345859013885,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2141650360","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011248907,0.0029310547,0.96354985,0.0009867058,0.00026499448,0.00013053807,0.00008793767,0.0005642583,0.020235665],"genre_scores_gemma":[0.76158506,0.0041096536,0.21567094,0.00055815245,0.00064109714,0.00052393484,0.00026286548,0.00017054162,0.016477743],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99799025,0.0006105948,0.00014087732,0.0004938099,0.0005847038,0.00017974565],"domain_scores_gemma":[0.9971209,0.0013586686,0.00044685017,0.00037880972,0.0005175074,0.0001771574],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019445495,0.001046143,0.0016133547,0.00072160095,0.000760858,0.0021409467,0.001443551,0.0013093019,0.004661807],"category_scores_gemma":[0.0069560874,0.00036665265,0.00086375023,0.0009924647,0.0019724136,0.00227215,0.0022892558,0.001596676,0.00071880413],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010517183,0.000115183604,0.0016336105,0.00032028448,0.00018444698,0.00019757113,0.00018936237,0.50310355,0.0018076341,0.3479144,0.005725288,0.13870354],"study_design_scores_gemma":[0.000045782308,0.000111100504,0.00046933824,0.00004612308,0.000044503122,0.00012232855,0.000041780702,0.6518942,0.00097146665,0.33186427,0.014360175,0.000028975333],"about_ca_topic_score_codex":0.0038723843,"about_ca_topic_score_gemma":0.0021501244,"teacher_disagreement_score":0.004661807,"about_ca_system_score_codex":0.0017055772,"about_ca_system_score_gemma":0.0013863228,"threshold_uncertainty_score":0.015595257},"labels":[],"label_agreement":null},{"id":"W2141690016","doi":"10.1145/1553374.1553383","title":"Predictive representations for policy gradient in POMDPs","year":2009,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Artificial intelligence; Mathematical optimization; Mathematics","score_opus":0.02090219095413405,"score_gpt":0.3127686912663465,"score_spread":0.29186650031221245,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2141690016","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017890578,0.00016466466,0.98030937,0.00017474801,0.000020984175,0.000034785266,0.000044274744,0.0002872457,0.0010734044],"genre_scores_gemma":[0.8373135,0.00032119208,0.1600858,0.00009475226,0.000056147474,0.00025132566,0.00018431728,0.00010548395,0.0015874965],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99936026,0.00023682084,0.000036404515,0.00013046325,0.00015956939,0.00007648349],"domain_scores_gemma":[0.9966048,0.0025516057,0.0003256861,0.00017930314,0.00025550864,0.000083047846],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021275035,0.0008519499,0.0013289434,0.0006742583,0.00037529672,0.0011745829,0.0012931827,0.0012398851,0.0020938343],"category_scores_gemma":[0.010883868,0.00067027035,0.00058621535,0.0006284362,0.0014013255,0.0021340237,0.0011683968,0.0018070757,0.0002394822],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000026636326,0.000011991811,0.0001574295,0.00003201066,0.0000083805,0.000019849256,0.000026669599,0.96869266,0.00017818746,0.021837663,0.000169054,0.008839514],"study_design_scores_gemma":[0.0000044804992,0.0000052400683,0.000018875951,0.0000034706222,0.0000015828911,0.000002696592,0.0000021445946,0.9916985,0.00009181257,0.008091482,0.00007770793,0.0000019929441],"about_ca_topic_score_codex":0.0051438627,"about_ca_topic_score_gemma":0.0028896898,"teacher_disagreement_score":0.0051438627,"about_ca_system_score_codex":0.0013126514,"about_ca_system_score_gemma":0.0012662399,"threshold_uncertainty_score":0.01125145},"labels":[],"label_agreement":null},{"id":"W2142394576","doi":"10.1109/tmech.2009.2024681","title":"Sequential $Q$-Learning With Kalman Filtering for Multirobot Cooperative Transportation","year":2009,"lang":"en","type":"article","venue":"IEEE/ASME Transactions on Mechatronics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Robot; Kalman filter; Artificial intelligence; Reinforcement learning; Sequence (biology); Table (database); Q-learning; Process (computing); Domain (mathematical analysis); Algorithm; Extended Kalman filter; Machine learning; Data mining; Mathematics","score_opus":0.018342717479080454,"score_gpt":0.25790694029586303,"score_spread":0.23956422281678258,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2142394576","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003345307,0.000101021156,0.9957989,0.000060225975,0.000021556945,0.000018822288,0.000004930993,0.00012352255,0.0005257005],"genre_scores_gemma":[0.6543241,0.00027132605,0.34220335,0.00014917682,0.000090305235,0.000236832,0.00005714184,0.000057780246,0.0026099554],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99882823,0.0004335799,0.00006852534,0.0002419633,0.00033504682,0.000092586044],"domain_scores_gemma":[0.9981432,0.0010509419,0.00019956549,0.00017348409,0.00036460726,0.000068299974],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023301123,0.00073906354,0.0009415115,0.0004392819,0.00058198266,0.0007465521,0.0015170389,0.001143198,0.0018885047],"category_scores_gemma":[0.0047678947,0.00042691015,0.0005683814,0.00060773356,0.0010848953,0.0015498844,0.0010670953,0.0011409693,0.0003424387],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011265278,0.00006756805,0.0006736129,0.000080145124,0.000050954663,0.00005970529,0.00009524824,0.8804791,0.0018187852,0.015945636,0.0006468719,0.09996966],"study_design_scores_gemma":[0.000013855842,0.00003326076,0.000077736135,0.0000033089234,0.000004891083,0.000009624462,0.0000046606146,0.9930996,0.00042110813,0.0057812403,0.0005458748,0.000004804589],"about_ca_topic_score_codex":0.005653037,"about_ca_topic_score_gemma":0.0033600752,"teacher_disagreement_score":0.005653037,"about_ca_system_score_codex":0.001187746,"about_ca_system_score_gemma":0.0015691947,"threshold_uncertainty_score":0.012322962},"labels":[],"label_agreement":null},{"id":"W2143680741","doi":"10.1109/ijcnn.2006.246661","title":"Aggregation of Reinforcement Learning Algorithms","year":2006,"lang":"en","type":"article","venue":"The 2006 IEEE International Joint Conference on Neural Network Proceedings","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Learning classifier system; Instance-based learning; Machine learning; Robustness (evolution); Robot learning; Online machine learning; Unsupervised learning; Algorithm; Proactive learning; Computational learning theory; Robot","score_opus":0.036895722843877536,"score_gpt":0.2604671221146379,"score_spread":0.22357139927076036,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2143680741","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023703977,0.00048698718,0.97117335,0.00014549006,0.000094679046,0.000114285904,0.000023414379,0.0008851622,0.0033727018],"genre_scores_gemma":[0.8213934,0.00036275142,0.17457603,0.0001416171,0.00009823667,0.00026740192,0.000084500396,0.000084343235,0.002991758],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983014,0.0005140507,0.00014724387,0.00035878157,0.0004923704,0.00018625482],"domain_scores_gemma":[0.9967891,0.0015150489,0.00033277384,0.00050072843,0.000690496,0.00017181494],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025820397,0.0011018872,0.001701934,0.00077609695,0.000488284,0.0011667013,0.0013716932,0.0008411498,0.002528564],"category_scores_gemma":[0.006537667,0.00042634743,0.0007347847,0.0006037819,0.00077736005,0.0013530654,0.0018220881,0.0014188024,0.00050857954],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001666998,0.00014445139,0.001624505,0.00014685946,0.00016274286,0.000106227846,0.00012516043,0.775465,0.002958669,0.014472526,0.0015192858,0.20310783],"study_design_scores_gemma":[0.000023575702,0.00008085779,0.00015453914,0.000010176751,0.000018676452,0.000032288113,0.000009376013,0.9906357,0.00090994686,0.0070216907,0.0010957337,0.0000074338127],"about_ca_topic_score_codex":0.0018565677,"about_ca_topic_score_gemma":0.0013819304,"teacher_disagreement_score":0.0025820397,"about_ca_system_score_codex":0.00081837835,"about_ca_system_score_gemma":0.0008688544,"threshold_uncertainty_score":0.013655305},"labels":[],"label_agreement":null},{"id":"W2144283793","doi":"","title":"Bounded Finite State Controllers","year":2003,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":162,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Bounded function; Mathematical optimization; Observable; Controller (irrigation); State space; Computer science; Mathematics; State (computer science); Vulnerability (computing); Applied mathematics; Control theory (sociology); Algorithm; Control (management); Artificial intelligence; Mathematical analysis","score_opus":0.012154928724894355,"score_gpt":0.22102379852353382,"score_spread":0.20886886979863947,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2144283793","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0014583826,0.00011742329,0.9949557,0.000050071078,0.000047540398,0.000023474222,0.000033246837,0.00082315644,0.0024910544],"genre_scores_gemma":[0.29921815,0.00048613528,0.6911644,0.00018163979,0.00006552948,0.00038849836,0.00030086722,0.00021815555,0.007976584],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99916065,0.00015224572,0.000039116716,0.00014995014,0.00041909498,0.00007894171],"domain_scores_gemma":[0.9990871,0.0004504045,0.00007860207,0.00016584276,0.00016192303,0.000056095832],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00068677956,0.00076682423,0.0008615481,0.00042215158,0.00044393097,0.0012993823,0.001917917,0.0010548256,0.0055752285],"category_scores_gemma":[0.0034182274,0.00037538545,0.00057357043,0.0004076343,0.0008000218,0.0014614641,0.0013790572,0.0016822217,0.0013547577],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000091871145,0.000059643884,0.00027600655,0.00013845075,0.000039732,0.00011479186,0.00007034459,0.75383437,0.004368441,0.108418934,0.003594144,0.12899321],"study_design_scores_gemma":[0.000009455257,0.00001614142,0.000016007789,0.000008560087,0.0000046782684,0.000017979464,0.00000248547,0.9850991,0.0011079959,0.009450977,0.004262158,0.000004547017],"about_ca_topic_score_codex":0.0030697654,"about_ca_topic_score_gemma":0.0034580706,"teacher_disagreement_score":0.0055752285,"about_ca_system_score_codex":0.0007866702,"about_ca_system_score_gemma":0.0015999486,"threshold_uncertainty_score":0.018651009},"labels":[],"label_agreement":null},{"id":"W2144794447","doi":"","title":"Bayes-Adaptive POMDPs","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":112,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval; McGill University","funders":"","keywords":"Partially observable Markov decision process; Reinforcement learning; Computer science; Markov decision process; Bellman equation; Artificial intelligence; Bayesian probability; Machine learning; Bayes' theorem; Markov process; Markov chain; Mathematical optimization; Markov model; Mathematics","score_opus":0.016486388323310792,"score_gpt":0.2496534226924639,"score_spread":0.23316703436915312,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2144794447","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00903169,0.00059288624,0.9813363,0.00053659594,0.00007873564,0.00009037509,0.00023377017,0.00028852012,0.0078111636],"genre_scores_gemma":[0.6138174,0.001532952,0.37294668,0.00032639984,0.00015957857,0.0005983475,0.0006304537,0.00010912141,0.009879159],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9981615,0.00064489665,0.00012480967,0.0003837694,0.0004983515,0.00018666577],"domain_scores_gemma":[0.99679995,0.0022149172,0.0003331369,0.00020003383,0.00030543288,0.00014650889],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025954084,0.001276553,0.0016560063,0.0007084392,0.00077862974,0.0017064031,0.0021202636,0.0016852195,0.0059302426],"category_scores_gemma":[0.008723672,0.0008309413,0.0012011514,0.00082193664,0.0018863067,0.002486642,0.0017511243,0.0026773724,0.0007026959],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000069814305,0.0000396915,0.0007856717,0.00013350313,0.000051449915,0.00010346086,0.0001207608,0.6881525,0.00039679592,0.28092834,0.0017032154,0.027514784],"study_design_scores_gemma":[0.000028356171,0.0000151067,0.0000879245,0.000015712048,0.00001064695,0.000018799408,0.000014134298,0.87008053,0.000122044185,0.12767613,0.0019203443,0.000010275206],"about_ca_topic_score_codex":0.008398677,"about_ca_topic_score_gemma":0.0070563005,"teacher_disagreement_score":0.008398677,"about_ca_system_score_codex":0.0021068228,"about_ca_system_score_gemma":0.0018307937,"threshold_uncertainty_score":0.019838631},"labels":[],"label_agreement":null},{"id":"W2144913588","doi":"10.1613/jair.2567","title":"Online Planning Algorithms for POMDPs","year":2008,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":515,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval; McGill University","funders":"National Institute of Mental Health; Fonds Québécois de la Recherche sur la Nature et les Technologies","keywords":"Partially observable Markov decision process; Computer science; Heuristic; Markov decision process; Mathematical optimization; Upper and lower bounds; Reduction (mathematics); Observable; Focus (optics); Computational complexity theory; Action (physics); Artificial intelligence; Machine learning; Markov process; Markov chain; Algorithm; Markov model; Mathematics","score_opus":0.4539481331437369,"score_gpt":0.4870388551056944,"score_spread":0.0330907219619575,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2144913588","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004665816,0.0005017859,0.99028075,0.00016238393,0.000044511995,0.00008982225,0.0000942896,0.00073446374,0.0034260848],"genre_scores_gemma":[0.38101903,0.001102506,0.61224765,0.00021318793,0.00009034325,0.0007896485,0.0005325382,0.000298679,0.0037064687],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988386,0.00040386117,0.00008412098,0.00024773387,0.00028267983,0.00014309569],"domain_scores_gemma":[0.99599034,0.0031983657,0.00027098515,0.00023390428,0.00018391655,0.00012249533],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018225992,0.0016927464,0.001766019,0.0008094936,0.0009390407,0.0014686141,0.0019484607,0.0015342613,0.0077229794],"category_scores_gemma":[0.00600829,0.00084955327,0.0011746348,0.0009942529,0.0014031102,0.0022789033,0.0021088899,0.0025227293,0.00083902973],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000077285054,0.000077930286,0.00024940525,0.00017988037,0.000042349,0.00005522608,0.00006714846,0.9046542,0.0003635836,0.046032075,0.0015248854,0.046676066],"study_design_scores_gemma":[0.000034775385,0.00002201183,0.000027405404,0.000017140494,0.000009255294,0.000013063463,0.000014349184,0.9623732,0.00022694943,0.0362148,0.0010414385,0.0000055623955],"about_ca_topic_score_codex":0.0064112013,"about_ca_topic_score_gemma":0.007155139,"teacher_disagreement_score":0.0077229794,"about_ca_system_score_codex":0.0016613695,"about_ca_system_score_gemma":0.0026899343,"threshold_uncertainty_score":0.025835931},"labels":[],"label_agreement":null},{"id":"W2149347537","doi":"10.1109/icsmc.2009.5346111","title":"A novel hybrid learning technique applied to a self-learning multi-robot system","year":2009,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Reinforcement learning; Computer science; Robot; Robot learning; Convergence (economics); Controller (irrigation); Fuzzy logic; Fuzzy control system; Artificial intelligence; Genetic algorithm; Robot control; Mobile robot; Control theory (sociology); Control (management); Machine learning","score_opus":0.01447734765549332,"score_gpt":0.2397579619282721,"score_spread":0.22528061427277876,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2149347537","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014633264,0.00029723835,0.9815175,0.0001345234,0.000054241664,0.000043007545,0.0000063671596,0.00026003463,0.0030539148],"genre_scores_gemma":[0.8170415,0.00032268048,0.17854482,0.00013210993,0.00005668003,0.00015977911,0.000018045588,0.000017967299,0.003706449],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99979633,0.00004647329,0.000011092933,0.000039099144,0.000087365006,0.000019702293],"domain_scores_gemma":[0.99981433,0.00006417633,0.000026523734,0.000022603203,0.000058769256,0.000013529799],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00038796605,0.00038056637,0.00044756662,0.00026703105,0.000348425,0.0004196781,0.00084636325,0.0008088975,0.0012706927],"category_scores_gemma":[0.0006250827,0.00015597789,0.0003734925,0.0002998578,0.0005119855,0.0004902528,0.00062094553,0.0004973126,0.00025115567],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011364393,0.000116401396,0.0008398639,0.00024866185,0.00010865977,0.0004568408,0.00031929617,0.6412402,0.046153773,0.03970145,0.0013900952,0.26931116],"study_design_scores_gemma":[0.000018225337,0.00010321793,0.000115017756,0.000007513307,0.0000096533895,0.00009372563,0.0000084013145,0.99271494,0.002161062,0.0030840333,0.001677039,0.0000071332315],"about_ca_topic_score_codex":0.0017637552,"about_ca_topic_score_gemma":0.0010726413,"teacher_disagreement_score":0.0017637552,"about_ca_system_score_codex":0.00033456934,"about_ca_system_score_gemma":0.000414155,"threshold_uncertainty_score":0.0042508245},"labels":[],"label_agreement":null},{"id":"W2149385746","doi":"","title":"Direct value-approximation for factored MDPs","year":2001,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":69,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Markov decision process; Bellman equation; Mathematical optimization; Linear programming; Linear approximation; Computation; Approximation error; Computer science; Approximation algorithm; Dynamic programming; Function (biology); Reduction (mathematics); Value (mathematics); Constraint (computer-aided design); Mathematics; Simple (philosophy); Applied mathematics; Markov process; Algorithm; Nonlinear system","score_opus":0.028321800899337682,"score_gpt":0.2669645503386068,"score_spread":0.2386427494392691,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2149385746","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006649993,0.00013718873,0.99179584,0.000051243795,0.000016547703,0.000035445206,0.00004369396,0.00024696047,0.0010231463],"genre_scores_gemma":[0.5059766,0.00038718482,0.49047682,0.000093581475,0.000050561473,0.00040829513,0.00027741675,0.00016964594,0.0021598246],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99924904,0.00023739324,0.000043654614,0.00014694493,0.00022272464,0.00010018529],"domain_scores_gemma":[0.99631625,0.0027856766,0.00022462137,0.00028595788,0.0002793486,0.00010814723],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017048329,0.0013143201,0.0019474659,0.000707898,0.00038820854,0.0012609,0.0013992719,0.001347368,0.004718922],"category_scores_gemma":[0.010493943,0.0007724179,0.001012915,0.0007068406,0.0012140092,0.0018724503,0.0013875299,0.001938633,0.00061574735],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006694447,0.000030376488,0.0002472404,0.00007629821,0.000022582246,0.00004625223,0.000040924948,0.9478051,0.00037117687,0.027023017,0.00047221157,0.023797844],"study_design_scores_gemma":[0.000011320049,0.000015785097,0.000015910844,0.000007780104,0.0000033072033,0.000006744095,0.0000034754814,0.9811335,0.0001774114,0.018378943,0.00024310812,0.0000027056574],"about_ca_topic_score_codex":0.0049010245,"about_ca_topic_score_gemma":0.0049382756,"teacher_disagreement_score":0.0049010245,"about_ca_system_score_codex":0.0014646231,"about_ca_system_score_gemma":0.0012966229,"threshold_uncertainty_score":0.01578641},"labels":[],"label_agreement":null},{"id":"W2149418961","doi":"10.1109/adprl.2007.368168","title":"Dual Representations for Dynamic Programming and Reinforcement Learning","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Temporal difference learning; Bellman equation; Dual (grammatical number); Dynamic programming; Computer science; Mathematical optimization; Divergence (linguistics); Representation (politics); Function (biology); Linear programming; Function approximation; Markov decision process; Exploit; Artificial intelligence; Mathematics; Artificial neural network; Markov process","score_opus":0.017369933620148447,"score_gpt":0.3033845091280638,"score_spread":0.28601457550791537,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2149418961","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0023902862,0.0002209456,0.99457693,0.00023283997,0.000040451218,0.00001853628,0.00002399107,0.000035038713,0.0024610257],"genre_scores_gemma":[0.44916645,0.0011848179,0.538186,0.00039496244,0.00030558382,0.0004937489,0.00021631198,0.00014544495,0.009906643],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9989225,0.0005202845,0.00005122448,0.00017562155,0.00022952711,0.00010087541],"domain_scores_gemma":[0.9980363,0.0012495779,0.00018781796,0.0001817376,0.00024368906,0.00010082792],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023520163,0.001040432,0.0011218217,0.00079811865,0.00040014525,0.0020452188,0.0012973691,0.0014066837,0.004090091],"category_scores_gemma":[0.007195348,0.0005066002,0.00089251646,0.0009951328,0.0016022002,0.002871015,0.0019534978,0.0033044883,0.00065963913],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004298906,0.00007375349,0.0003106463,0.000120602715,0.000030308029,0.000041380674,0.000076634096,0.23081681,0.0007232404,0.7306567,0.0013928318,0.035714168],"study_design_scores_gemma":[0.000012671394,0.00002970603,0.000038188722,0.000015750264,0.000006982062,0.000023216106,0.000010835421,0.7865827,0.00036274787,0.21076284,0.002147436,0.0000069441985],"about_ca_topic_score_codex":0.0010237986,"about_ca_topic_score_gemma":0.0007747562,"teacher_disagreement_score":0.004090091,"about_ca_system_score_codex":0.0015751004,"about_ca_system_score_gemma":0.0013587375,"threshold_uncertainty_score":0.013682723},"labels":[],"label_agreement":null},{"id":"W2149586740","doi":"10.1613/jair.4117","title":"Scalable and Efficient Bayes-Adaptive Reinforcement Learning Based on Monte-Carlo Tree Search","year":2013,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":79,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Gatsby Charitable Foundation; Royal Society","keywords":"Computer science; Monte Carlo tree search; Machine learning; Reinforcement learning; Tree (set theory); Scalability; Bayesian probability; Artificial intelligence; Benchmark (surveying); Monte Carlo method; Mathematical optimization; Mathematics; Statistics","score_opus":0.1097453708321637,"score_gpt":0.354419875162297,"score_spread":0.24467450433013332,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2149586740","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0108155515,0.00018212954,0.98611635,0.00017237695,0.00002482189,0.00006026662,0.00003358808,0.00059852994,0.0019963046],"genre_scores_gemma":[0.6410341,0.00020209915,0.35623515,0.00019020581,0.000046593093,0.00030668985,0.00014835056,0.00014845926,0.0016882974],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99882764,0.00045421819,0.00004834862,0.00015082721,0.00040418195,0.000114712006],"domain_scores_gemma":[0.9960204,0.003024276,0.00025300603,0.00026495164,0.00027938018,0.00015805483],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019268566,0.00075084256,0.0017972225,0.0005669168,0.0004505221,0.0008539415,0.0019407878,0.0011824883,0.002998225],"category_scores_gemma":[0.00971664,0.00058291474,0.0005686133,0.0006918503,0.0012391879,0.0014785441,0.0015748439,0.0020575214,0.00053894636],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011667205,0.000086725384,0.00075426575,0.00006337798,0.000040152863,0.000054620647,0.00007465323,0.90034837,0.0007350576,0.02501696,0.0014431366,0.07126589],"study_design_scores_gemma":[0.0000110572755,0.000009978257,0.0000296098,0.0000032515927,0.00000262589,0.0000058580094,0.000002460069,0.9931479,0.000085801206,0.0065672677,0.00013184991,0.0000022895927],"about_ca_topic_score_codex":0.0068948865,"about_ca_topic_score_gemma":0.008601874,"teacher_disagreement_score":0.0068948865,"about_ca_system_score_codex":0.0011651064,"about_ca_system_score_gemma":0.0026343225,"threshold_uncertainty_score":0.0137094855},"labels":[],"label_agreement":null},{"id":"W2150468603","doi":"10.1613/jair.3912","title":"The Arcade Learning Environment: An Evaluation Platform for General Agents","year":2013,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1060,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alberta Innovates; University of Alberta; Compute Canada","keywords":"Testbed; Benchmarking; Reinforcement learning; Benchmark (surveying); Imitation; Interface (matter)","score_opus":0.29025970311617355,"score_gpt":0.4427010381262976,"score_spread":0.15244133501012402,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2150468603","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07411522,0.0013576704,0.79534334,0.0016742568,0.0007121144,0.0049086013,0.008251042,0.08102206,0.03261568],"genre_scores_gemma":[0.27565315,0.000527038,0.6941121,0.0006190911,0.000090828566,0.007379126,0.009865311,0.0069187954,0.004834517],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98879224,0.0073316116,0.00080970867,0.00082199235,0.0018166933,0.00042776004],"domain_scores_gemma":[0.9787911,0.0133022545,0.0009883412,0.0029910083,0.0028807265,0.0010465299],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013376182,0.0019080254,0.0010788306,0.0022869513,0.00058522413,0.001953681,0.0038308182,0.0017376447,0.0084892195],"category_scores_gemma":[0.034813408,0.0007657617,0.0009738894,0.0013274633,0.0012555573,0.0028622579,0.003909424,0.0031130782,0.0029362289],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0035493956,0.003639912,0.0098472,0.0033191657,0.00091482897,0.000463411,0.0011178679,0.43748447,0.007879542,0.050999735,0.14447126,0.33631325],"study_design_scores_gemma":[0.0014943657,0.0020208918,0.0023269684,0.00025040036,0.00012467943,0.00018657433,0.00023847545,0.8897074,0.00915134,0.023755107,0.07060668,0.00013715627],"about_ca_topic_score_codex":0.0047468194,"about_ca_topic_score_gemma":0.004241815,"teacher_disagreement_score":0.013376182,"about_ca_system_score_codex":0.001480088,"about_ca_system_score_gemma":0.0023186465,"threshold_uncertainty_score":0.07074082},"labels":[],"label_agreement":null},{"id":"W2150526575","doi":"10.1109/tai.2002.1180840","title":"Reinforcement learning in multiagent systems: a modular fuzzy approach with internal model capabilities","year":2003,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Reinforcement learning; Modular design; Computer science; Robustness (evolution); Fuzzy logic; Artificial intelligence; Multi-agent system; Fuzzy rule; Fuzzy control system; State space; Mathematics","score_opus":0.019054210749540022,"score_gpt":0.22208761357715973,"score_spread":0.2030334028276197,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2150526575","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008891517,0.00025372376,0.9889664,0.0001787681,0.00002268262,0.0000235556,0.0000067547935,0.00013839576,0.0015182338],"genre_scores_gemma":[0.75274336,0.00044466546,0.24467438,0.000093535426,0.00009076257,0.00014852185,0.000025474508,0.000036117977,0.0017432532],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996313,0.0001422872,0.000018960702,0.000056626664,0.00012178446,0.000029030587],"domain_scores_gemma":[0.9994904,0.00022642354,0.000075662865,0.00007203948,0.00009498914,0.000040556228],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010498081,0.0005411614,0.00064671924,0.00033241155,0.0003014978,0.00075735786,0.0011966377,0.0008287497,0.0011248088],"category_scores_gemma":[0.0018469011,0.00026765093,0.0006769545,0.00025733156,0.0009525997,0.001268144,0.0011486948,0.0010371705,0.0002047711],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006634742,0.00006152893,0.00045871234,0.0001188771,0.0000834421,0.00019203885,0.00019458645,0.8201821,0.006165654,0.09369832,0.000643122,0.0781354],"study_design_scores_gemma":[0.000017883513,0.00004192812,0.000066727385,0.000009800647,0.000011126111,0.000026692724,0.000006961347,0.9682667,0.0007695441,0.03001699,0.0007576983,0.000007880359],"about_ca_topic_score_codex":0.001352011,"about_ca_topic_score_gemma":0.0009274981,"teacher_disagreement_score":0.001352011,"about_ca_system_score_codex":0.0005629194,"about_ca_system_score_gemma":0.0004622529,"threshold_uncertainty_score":0.0055519342},"labels":[],"label_agreement":null},{"id":"W2151416233","doi":"","title":"Fitted Q-iteration in continuous action-space MDPs","year":2007,"lang":"en","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":149,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Action (physics); Q-learning; Bellman equation; Trajectory; Set (abstract data type); Mathematical optimization; Action selection; Markov decision process; State space; Computer science; State (computer science); Function (biology); Value (mathematics); Space (punctuation); Selection (genetic algorithm); Mathematics; Algorithm; Artificial intelligence; Markov process; Machine learning; Statistics","score_opus":0.013775759560532399,"score_gpt":0.24041286006998908,"score_spread":0.22663710050945668,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2151416233","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03745475,0.0002183296,0.9604027,0.00020495642,0.000029308505,0.000049920673,0.000034152334,0.00021898563,0.0013869057],"genre_scores_gemma":[0.81100017,0.00021600201,0.18514162,0.00013201735,0.000039878236,0.00027023847,0.00012264143,0.00010109724,0.0029763067],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99815685,0.00097131595,0.00007051754,0.00027672603,0.00033547907,0.00018908115],"domain_scores_gemma":[0.9884686,0.009002777,0.00071307906,0.000454944,0.00079761934,0.00056291244],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043338365,0.0009971182,0.0017943803,0.0005977907,0.0004997684,0.0012711182,0.0019361997,0.0021948665,0.0022049642],"category_scores_gemma":[0.01910344,0.00090822874,0.0008128828,0.00069623365,0.003256886,0.0022016314,0.0015770535,0.002137963,0.00039598925],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011743171,0.000039506725,0.0003652794,0.000041819523,0.000024329456,0.00005630873,0.000051203282,0.9681436,0.00032554098,0.025432998,0.00020276914,0.0051991367],"study_design_scores_gemma":[0.000016793916,0.00003165851,0.000027934087,0.0000042771403,0.0000024810035,0.000006165576,0.0000045232946,0.9901957,0.000111859925,0.009494732,0.00010043348,0.0000033569195],"about_ca_topic_score_codex":0.0047861645,"about_ca_topic_score_gemma":0.0025172792,"teacher_disagreement_score":0.0047861645,"about_ca_system_score_codex":0.0017239699,"about_ca_system_score_gemma":0.0016684866,"threshold_uncertainty_score":0.022919834},"labels":[],"label_agreement":null},{"id":"W2151620419","doi":"10.1111/coin.12016","title":"EFFICIENT ABSTRACTION SELECTION IN REINFORCEMENT LEARNING","year":2013,"lang":"en","type":"article","venue":"Computational Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Abstraction; Markov decision process; Action selection; Selection (genetic algorithm); Context (archaeology); Artificial intelligence; Set (abstract data type); Machine learning; State space; Class (philosophy); Theoretical computer science; Markov process; Programming language; Mathematics; Agency (philosophy)","score_opus":0.02106858788774564,"score_gpt":0.2715625062281282,"score_spread":0.2504939183403826,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2151620419","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07159132,0.00034262796,0.92529553,0.00025877447,0.000022608054,0.00009772843,0.000035419267,0.00025702282,0.0020990246],"genre_scores_gemma":[0.90623736,0.0001688443,0.09230384,0.00007420995,0.000020736448,0.00016885299,0.00005637568,0.000032752694,0.0009369577],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985759,0.00071691565,0.000061986124,0.00022887837,0.00024654483,0.00016971547],"domain_scores_gemma":[0.9973598,0.0018719292,0.0002685065,0.00020533362,0.00015546544,0.0001389663],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023697838,0.00092827826,0.0013990913,0.00045758823,0.00047609367,0.00077605486,0.0009923677,0.0010628044,0.0013642632],"category_scores_gemma":[0.0070790723,0.00051137386,0.00051678653,0.00040970315,0.0017613382,0.0016288854,0.0018033114,0.0013833717,0.00018093886],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010584891,0.000053404467,0.00088153145,0.00006169309,0.000038892726,0.00007041125,0.000087085544,0.9473156,0.0010006747,0.026937572,0.0003175611,0.023129828],"study_design_scores_gemma":[0.000025024245,0.000036079717,0.000073677555,0.0000071082045,0.000008438722,0.000009686509,0.000009375013,0.9799296,0.00040519526,0.019253517,0.0002377506,0.000004608445],"about_ca_topic_score_codex":0.0021822306,"about_ca_topic_score_gemma":0.0017806075,"teacher_disagreement_score":0.0023697838,"about_ca_system_score_codex":0.001207291,"about_ca_system_score_gemma":0.0012542974,"threshold_uncertainty_score":0.012532711},"labels":[],"label_agreement":null},{"id":"W2152469292","doi":"10.3166/ria.27.171-194","title":"Stratégies d’échantillonnage pour l’apprentissage par renforcement batch","year":2013,"lang":"fr","type":"article","venue":"Revue d intelligence artificielle","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Psychology","score_opus":0.051993114263749474,"score_gpt":0.2708318065741388,"score_spread":0.21883869231038933,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2152469292","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05587986,0.0008132286,0.9271196,0.0014470301,0.0005241438,0.00040566688,0.00013972408,0.0020455252,0.011625241],"genre_scores_gemma":[0.7097182,0.0005808637,0.2371596,0.0004362178,0.00034720052,0.00065662584,0.00031517813,0.0005162834,0.05026988],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9954391,0.001829975,0.00021206745,0.0007562673,0.001250267,0.0005123355],"domain_scores_gemma":[0.9846705,0.009755468,0.0005658896,0.0014903136,0.0027066353,0.00081117556],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052741347,0.0014053996,0.002008208,0.001067939,0.0015588069,0.002939806,0.0035598879,0.003249812,0.015069825],"category_scores_gemma":[0.02074962,0.0007408681,0.00096034136,0.00058565574,0.0017509918,0.0031183523,0.002155736,0.00317747,0.003322938],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021076004,0.00064078876,0.002398769,0.0004722652,0.00020929879,0.0008874345,0.0006749006,0.4971136,0.01772863,0.11608534,0.019597495,0.3420839],"study_design_scores_gemma":[0.00013172586,0.00017901052,0.0005893854,0.00003193995,0.000030846335,0.00013236204,0.000063796164,0.9546283,0.005870114,0.031773753,0.0065134107,0.000055307733],"about_ca_topic_score_codex":0.0107673155,"about_ca_topic_score_gemma":0.009884753,"teacher_disagreement_score":0.015069825,"about_ca_system_score_codex":0.0018192009,"about_ca_system_score_gemma":0.0041939155,"threshold_uncertainty_score":0.05041355},"labels":[],"label_agreement":null},{"id":"W2152711271","doi":"10.1109/acc.2011.5990832","title":"Decentralized learning in two-player zero-sum games: A L&lt;inf&gt;R-I&lt;/inf&gt; lagging anchor algorithm","year":2011,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Lagging; Zero (linguistics); Algorithm; Action (physics); Computer science; Zero-sum game; Matrix (chemical analysis); Nash equilibrium; Artificial intelligence; Mathematics; Discrete mathematics; Mathematical optimization; Statistics; Physics; Philosophy","score_opus":0.025089321999201167,"score_gpt":0.26151948006262016,"score_spread":0.236430158063419,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2152711271","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0081035765,0.000058104008,0.9893568,0.00016734487,0.000023658282,0.000045088225,0.000016671876,0.00020473862,0.0020240834],"genre_scores_gemma":[0.5637689,0.00022951752,0.42696902,0.00024120591,0.00008746674,0.00044054692,0.00013865888,0.0001302745,0.007994421],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99878305,0.00045123958,0.00005147215,0.00026542266,0.00032071455,0.000128092],"domain_scores_gemma":[0.9974124,0.0015650741,0.00023679108,0.0002872488,0.00030493873,0.00019352329],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002331496,0.0008877749,0.0016629827,0.00048296226,0.00072623836,0.0010607985,0.0021680375,0.0015672313,0.0041946066],"category_scores_gemma":[0.006525401,0.0004724557,0.00060041476,0.0007877208,0.0014860781,0.001901245,0.0022723998,0.0024152764,0.0009078801],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001590353,0.00023902961,0.00056747143,0.000094514464,0.000044651344,0.00008063837,0.00012437276,0.78416556,0.0017604115,0.111404724,0.0025561177,0.098803505],"study_design_scores_gemma":[0.000065243396,0.00006301374,0.000043079843,0.00000612585,0.000005459143,0.000015685915,0.0000083073855,0.9693522,0.0004363715,0.029371621,0.00062457635,0.000008269689],"about_ca_topic_score_codex":0.0025865692,"about_ca_topic_score_gemma":0.0026478462,"teacher_disagreement_score":0.0041946066,"about_ca_system_score_codex":0.0011753892,"about_ca_system_score_gemma":0.002229027,"threshold_uncertainty_score":0.014032364},"labels":[],"label_agreement":null},{"id":"W2153516274","doi":"10.5555/2772879.2773311","title":"Incremental Policy Iteration with Guaranteed Escape from Local Optima in POMDP Planning","year":2015,"lang":"en","type":"article","venue":"Kent Academic Repository (University of Kent)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Partially observable Markov decision process; Mathematical optimization; Markov decision process; Local optimum; Bounded function; Controller (irrigation); Local search (optimization); State (computer science); Markov chain; Markov process; Artificial intelligence; Markov model; Algorithm; Mathematics; Machine learning","score_opus":0.019518381185233375,"score_gpt":0.22832623942463323,"score_spread":0.20880785823939985,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2153516274","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022342943,0.00030637835,0.9715505,0.00029047485,0.000035850888,0.000087762885,0.00004645546,0.00074564025,0.0045940657],"genre_scores_gemma":[0.79430866,0.00025627756,0.20149674,0.0002385123,0.000039559513,0.0004917534,0.00013136887,0.00020047397,0.0028366153],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986124,0.00060909527,0.00006899555,0.00023654975,0.00030156254,0.00017131367],"domain_scores_gemma":[0.99373937,0.005213795,0.00033912074,0.00023615702,0.00026373926,0.00020782847],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025663187,0.0013272054,0.0019295554,0.00075254816,0.00063773256,0.0010968889,0.0016272677,0.0014666839,0.002671133],"category_scores_gemma":[0.008507795,0.00090258685,0.0009381459,0.000587251,0.0024564865,0.0015113004,0.002293265,0.0025391853,0.00037490533],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007229197,0.00004368364,0.00026224816,0.00006525712,0.000028793631,0.000057468125,0.00007896466,0.9718622,0.00030405974,0.014448781,0.00047900976,0.012297251],"study_design_scores_gemma":[0.000017745593,0.000024272613,0.00002504927,0.000007597071,0.000004788069,0.0000071137456,0.000006671844,0.9907691,0.00015649019,0.008788875,0.00018804788,0.000004335907],"about_ca_topic_score_codex":0.0062497174,"about_ca_topic_score_gemma":0.0061930986,"teacher_disagreement_score":0.0062497174,"about_ca_system_score_codex":0.0014377383,"about_ca_system_score_gemma":0.002696638,"threshold_uncertainty_score":0.013572156},"labels":[],"label_agreement":null},{"id":"W2156760524","doi":"10.1109/ijcnn.1999.833417","title":"Adaptive exploration in reinforcement learning","year":2003,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph; University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Implementation; Artificial intelligence; Connectionism; Machine learning; Reinforcement; Learning classifier system; Artificial neural network; Engineering; Software engineering","score_opus":0.036640872771695775,"score_gpt":0.2510125592442868,"score_spread":0.214371686472591,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2156760524","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011953853,0.0017520059,0.9760165,0.00058025296,0.00011038491,0.000044011722,0.00002572855,0.00021600489,0.009301226],"genre_scores_gemma":[0.84152937,0.0018727304,0.14902583,0.0002874199,0.00019023928,0.00033949484,0.000057383364,0.0000623718,0.006635262],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99934965,0.0003157757,0.00003337494,0.00009577528,0.00014859339,0.000056925684],"domain_scores_gemma":[0.99849725,0.0010792708,0.0001200674,0.00008465056,0.00014891503,0.000069873124],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013025742,0.0006420457,0.0007314996,0.00030637975,0.00034499392,0.0009865909,0.0009043261,0.0010554739,0.0021111118],"category_scores_gemma":[0.004919053,0.00028103395,0.0003741045,0.00037847064,0.0017292465,0.0012716843,0.0010089792,0.0014616123,0.00032647347],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000080759324,0.00005936024,0.0006732265,0.00016798495,0.00007409656,0.00012169516,0.0001482305,0.65815055,0.0013721474,0.27956942,0.0018515224,0.057730976],"study_design_scores_gemma":[0.00004673173,0.000049796596,0.000104801315,0.000018447441,0.000011921279,0.000030046189,0.000012583121,0.82385516,0.00047312243,0.17278512,0.0026010375,0.000011206099],"about_ca_topic_score_codex":0.0017504335,"about_ca_topic_score_gemma":0.0011545472,"teacher_disagreement_score":0.0021111118,"about_ca_system_score_codex":0.00097383786,"about_ca_system_score_gemma":0.0007336767,"threshold_uncertainty_score":0.0070657134},"labels":[],"label_agreement":null},{"id":"W2157544132","doi":"10.48550/arxiv.1207.5554","title":"Bellman Error Based Feature Generation using Random Projections on Sparse Spaces","year":2012,"lang":"en","type":"article","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Bellman equation; Computer science; Logarithm; Dimension (graph theory); Sparse approximation; Contraction (grammar); Representation (politics); Mathematical optimization; Algorithm; Feature (linguistics); Convergence (economics); Mathematics; Artificial intelligence","score_opus":0.03420690341322675,"score_gpt":0.2673350574748708,"score_spread":0.23312815406164406,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2157544132","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009653842,0.000046656667,0.98950094,0.0000774588,0.000012952221,0.000038737508,0.000022850532,0.00024347044,0.00040298616],"genre_scores_gemma":[0.535874,0.000092105794,0.461627,0.000115021954,0.000037248225,0.00034524297,0.00016286883,0.00013722047,0.0016092539],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983506,0.00070794695,0.00006967017,0.00025930998,0.00046524475,0.00014729706],"domain_scores_gemma":[0.99369544,0.0040354347,0.0005989361,0.0007822252,0.0007213788,0.00016659524],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002931492,0.00081670267,0.0013801693,0.0006843465,0.00049768266,0.00084267725,0.0013784316,0.0010679028,0.0022576484],"category_scores_gemma":[0.012357381,0.0005553018,0.0007271577,0.0006827657,0.0013585311,0.0019895837,0.0019255768,0.0018698678,0.0004630871],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023068865,0.000116581316,0.00093430857,0.000091389906,0.00004876127,0.000095936324,0.00011898048,0.8066194,0.0039939517,0.048268236,0.0012715742,0.13821015],"study_design_scores_gemma":[0.000011964592,0.000038171485,0.00004964419,0.000004773627,0.0000027870037,0.000011324121,0.0000028201734,0.9898921,0.0010344414,0.008762934,0.0001837586,0.000005250269],"about_ca_topic_score_codex":0.0020809725,"about_ca_topic_score_gemma":0.0018915982,"teacher_disagreement_score":0.002931492,"about_ca_system_score_codex":0.00093128893,"about_ca_system_score_gemma":0.0012631778,"threshold_uncertainty_score":0.0155034065},"labels":[],"label_agreement":null},{"id":"W2157864803","doi":"","title":"Action-Gap Phenomenon in Reinforcement Learning","year":2011,"lang":"en","type":"article","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Action (physics); Phenomenon; Bounded function; Q-learning; Bellman equation; Function (biology); Value (mathematics); Mathematics; Reinforcement; Computer science; Mathematical optimization; Artificial intelligence; Statistics; Mathematical analysis; Physics; Engineering; Structural engineering","score_opus":0.03411165688407607,"score_gpt":0.24838523452784825,"score_spread":0.21427357764377217,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2157864803","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.078308105,0.00084044,0.9123435,0.0010395809,0.00007547034,0.00006788461,0.00006209844,0.00044326865,0.00681966],"genre_scores_gemma":[0.9538204,0.00028359788,0.043764353,0.00018276942,0.00006965299,0.00011345674,0.000046489826,0.00009510979,0.0016242057],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99759126,0.0011845044,0.0000941019,0.00037676582,0.00051991775,0.00023345771],"domain_scores_gemma":[0.97359776,0.021458926,0.0018437976,0.0017409078,0.00075950543,0.000599084],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005384113,0.00074962823,0.001213655,0.0005244869,0.0005708884,0.0013633526,0.0010223385,0.0014677431,0.0022124029],"category_scores_gemma":[0.03425872,0.00041707375,0.0006995895,0.00038224272,0.004385935,0.0032205794,0.002333862,0.0032709031,0.0002768243],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005757967,0.00011380363,0.0049193655,0.0003421797,0.00010907346,0.00035983467,0.0004881518,0.5558712,0.0050773527,0.40256765,0.0015452722,0.02803045],"study_design_scores_gemma":[0.000039432467,0.0002228537,0.00066454266,0.000029360337,0.0000129271575,0.00007078761,0.000032075433,0.8819094,0.0012198561,0.114967786,0.0008145965,0.000016423655],"about_ca_topic_score_codex":0.001179298,"about_ca_topic_score_gemma":0.00052952993,"teacher_disagreement_score":0.005384113,"about_ca_system_score_codex":0.0013157856,"about_ca_system_score_gemma":0.000895878,"threshold_uncertainty_score":0.028474271},"labels":[],"label_agreement":null},{"id":"W2158304715","doi":"10.1109/tsmcb.2007.899419","title":"Positive Impact of State Similarity on Reinforcement Learning Performance","year":2007,"lang":"en","type":"article","venue":"IEEE Transactions on Systems Man and Cybernetics Part B (Cybernetics)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Reinforcement learning; Similarity (geometry); Artificial intelligence; Context (archaeology); Computer science; Reinforcement; Function (biology); Bellman equation; Tree (set theory); State (computer science); Action (physics); Machine learning; Value (mathematics); Mathematics; Mathematical optimization; Algorithm; Engineering","score_opus":0.01715931351416537,"score_gpt":0.2565818502936981,"score_spread":0.23942253677953276,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2158304715","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6549893,0.0010058216,0.33172902,0.0009371417,0.00018184049,0.00016369538,0.00007587703,0.0014464505,0.009470782],"genre_scores_gemma":[0.989744,0.000058373895,0.009695958,0.0000621413,0.000030889347,0.000025590542,0.000039411138,0.000032297998,0.00031135548],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99583924,0.001362264,0.0003194651,0.00081167987,0.0013024558,0.00036501093],"domain_scores_gemma":[0.9497266,0.039184302,0.0036691786,0.0034781334,0.0024721448,0.0014695657],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005013166,0.0008216931,0.0014841888,0.00061307504,0.0006221985,0.0014097736,0.0009282367,0.0014368062,0.0021725914],"category_scores_gemma":[0.050867554,0.00028993684,0.0003792305,0.00040156025,0.0013536621,0.0030769506,0.0019297915,0.0020953533,0.0003985104],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023011733,0.0020039116,0.03193978,0.0004015616,0.00040386617,0.00041945904,0.00034932693,0.594425,0.02118233,0.023799218,0.0014612285,0.32131314],"study_design_scores_gemma":[0.00010436873,0.0016720615,0.011001276,0.000032266504,0.00007951412,0.00023877306,0.00009125263,0.95402586,0.009196662,0.022914404,0.000596108,0.000047499252],"about_ca_topic_score_codex":0.0010803441,"about_ca_topic_score_gemma":0.0009696226,"teacher_disagreement_score":0.005013166,"about_ca_system_score_codex":0.0008439706,"about_ca_system_score_gemma":0.0011852283,"threshold_uncertainty_score":0.026512504},"labels":[],"label_agreement":null},{"id":"W2158738729","doi":"","title":"Regularized Policy Iteration","year":2008,"lang":"en","type":"article","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":108,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; University of Alberta","funders":"","keywords":"Reinforcement learning; Mathematical optimization; Regularization (linguistics); Reproducing kernel Hilbert space; Temporal difference learning; Computer science; Bellman equation; Markov decision process; Kernel (algebra); Residual; Hilbert space; Parametric statistics; Convergence (economics); Function approximation; Rate of convergence; Algorithm; Mathematics; Artificial intelligence; Markov process; Artificial neural network","score_opus":0.013613790347172965,"score_gpt":0.2341841148069613,"score_spread":0.22057032445978833,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2158738729","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005806307,0.00017465679,0.9914238,0.00014406595,0.000049511294,0.000037553404,0.000020700783,0.00012717451,0.0022161896],"genre_scores_gemma":[0.6361047,0.0003598688,0.3534197,0.000308216,0.00010460431,0.00038650987,0.00014400594,0.00015060752,0.00902177],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99874437,0.0005407333,0.00005282442,0.0002464064,0.00030604968,0.00010953901],"domain_scores_gemma":[0.99704725,0.0018626776,0.00026375076,0.00027510064,0.000442888,0.00010836725],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019029903,0.00089245057,0.0013496778,0.00036923235,0.00033622695,0.0012032688,0.001278156,0.0015182587,0.0032464187],"category_scores_gemma":[0.007979701,0.00042792916,0.00048330607,0.0003768031,0.0016634316,0.0014306376,0.0013518037,0.0014377641,0.0006724038],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010062351,0.000054942397,0.00035139415,0.000100246805,0.00004240359,0.00005687468,0.000068178124,0.8846641,0.0011877465,0.0785394,0.0010011032,0.033832952],"study_design_scores_gemma":[0.000009769747,0.000020043542,0.000017842862,0.000005249155,0.0000026680411,0.000009823469,0.0000032111534,0.9895075,0.00030205637,0.009545557,0.0005728809,0.0000034619866],"about_ca_topic_score_codex":0.001994019,"about_ca_topic_score_gemma":0.001146782,"teacher_disagreement_score":0.0032464187,"about_ca_system_score_codex":0.0010800379,"about_ca_system_score_gemma":0.0017806066,"threshold_uncertainty_score":0.0108603835},"labels":[],"label_agreement":null},{"id":"W2159008857","doi":"","title":"Learning to Control an Octopus Arm with Gaussian Process Temporal Difference Methods","year":2005,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"octopus (software); Gaussian process; Robotic arm; Reinforcement learning; Temporal difference learning; Computer science; Artificial intelligence; Process (computing); Domain (mathematical analysis); Bayesian optimization; Bayesian probability; State space; Trajectory; Machine learning; Computer vision; Gaussian; Mathematics","score_opus":0.020160526124716367,"score_gpt":0.3137583755310238,"score_spread":0.29359784940630745,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2159008857","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040192652,0.00011991111,0.95766467,0.00021626794,0.000030133158,0.000034055094,0.0000120454315,0.00019359971,0.0015365572],"genre_scores_gemma":[0.8975414,0.00008964018,0.10008913,0.00010510166,0.000017327598,0.00010302683,0.000028429331,0.000025196194,0.0020006618],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99975437,0.00008929439,0.000015219017,0.00004316212,0.00006688849,0.000031053336],"domain_scores_gemma":[0.99871564,0.00090084557,0.00013324684,0.00004799545,0.00015122624,0.000051097322],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012483407,0.00042853903,0.00056533684,0.00024216516,0.00027562605,0.0005227871,0.0005831323,0.00075153727,0.0009799373],"category_scores_gemma":[0.0029689146,0.00028587395,0.00037608482,0.00023307206,0.00071228476,0.0004964898,0.0006865066,0.0009550925,0.00012409785],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005754593,0.000034207784,0.00025543245,0.000021064176,0.000016266069,0.000023657627,0.000040161238,0.97394764,0.0011065394,0.005065539,0.00020198374,0.019229915],"study_design_scores_gemma":[0.000004828745,0.000009906638,0.000015532107,8.678791e-7,0.0000010967599,0.0000015445789,8.879275e-7,0.9991431,0.00013737337,0.00063533103,0.000048229947,0.0000012825052],"about_ca_topic_score_codex":0.0082133515,"about_ca_topic_score_gemma":0.004601019,"teacher_disagreement_score":0.0082133515,"about_ca_system_score_codex":0.0007421789,"about_ca_system_score_gemma":0.0009401019,"threshold_uncertainty_score":0.016331077},"labels":[],"label_agreement":null},{"id":"W2159136633","doi":"10.5555/2627435.2750354","title":"Efficient learning and planning with compressed predictive states","year":2014,"lang":"en","type":"article","venue":"Journal of Machine Learning Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Reinforcement learning; Artificial intelligence; Observable; A priori and a posteriori; Dimensionality reduction; Machine learning; Curse of dimensionality","score_opus":0.0249183959460578,"score_gpt":0.331304983517485,"score_spread":0.30638658757142717,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2159136633","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009767411,0.00014130655,0.9876452,0.00025378334,0.000021618354,0.00003966036,0.0001056526,0.0004894075,0.0015360288],"genre_scores_gemma":[0.6970861,0.0003666247,0.29874808,0.00021754841,0.00007314551,0.00031320337,0.0005853173,0.00017686392,0.0024331363],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99935013,0.00022210804,0.000035641755,0.00013563693,0.00018730988,0.00006918975],"domain_scores_gemma":[0.9969933,0.0021859906,0.00026075382,0.0002964257,0.0001880986,0.00007538916],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010261993,0.0009278636,0.0010285862,0.00053637195,0.00037464662,0.00094837503,0.0013524804,0.0011043325,0.0024051366],"category_scores_gemma":[0.006623545,0.0007307275,0.00063332065,0.0006720937,0.0015036294,0.002236263,0.0017753841,0.0022095933,0.00029672674],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000059290756,0.00002287417,0.00018136675,0.000048825088,0.000013479001,0.000048991882,0.000064559725,0.945515,0.0006830539,0.031121122,0.0007127859,0.021528732],"study_design_scores_gemma":[0.0000050639037,0.00000817096,0.000018639856,0.000004436293,0.000001732015,0.0000056085987,0.000004447311,0.9857195,0.00023637226,0.013818709,0.00017470968,0.0000026621199],"about_ca_topic_score_codex":0.0067225057,"about_ca_topic_score_gemma":0.006948614,"teacher_disagreement_score":0.0067225057,"about_ca_system_score_codex":0.0011131369,"about_ca_system_score_gemma":0.0016629606,"threshold_uncertainty_score":0.013366759},"labels":[],"label_agreement":null},{"id":"W2159272820","doi":"10.1613/jair.3175","title":"Non-Deterministic Policies in Markovian Decision Processes","year":2011,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"National Institutes of Health; Duke-NUS Medical School; Natural Sciences and Engineering Research Council of Canada; McGill University","keywords":"Reinforcement learning; Computer science; Flexibility (engineering); Markov decision process; Set (abstract data type); Task (project management); Action selection; Markov process; Construct (python library); Process (computing); Action (physics); Decision problem; Artificial intelligence; Selection (genetic algorithm); Decision support system; Machine learning; Operations research; Algorithm; Mathematics; Engineering","score_opus":0.23871200327016803,"score_gpt":0.42671615641768834,"score_spread":0.1880041531475203,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2159272820","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027043002,0.0003180952,0.9706632,0.00031529574,0.0000303063,0.00005109408,0.000057237557,0.00020862345,0.0013131668],"genre_scores_gemma":[0.7207313,0.0007359307,0.27583215,0.00016276893,0.000057685444,0.00032162698,0.00016319187,0.000062688414,0.0019326775],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998049,0.0010928058,0.00011156658,0.00028293274,0.00028947028,0.00017420678],"domain_scores_gemma":[0.9877359,0.010646513,0.0007132209,0.00033097388,0.00034397547,0.00022951925],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003230904,0.00085937744,0.0011229401,0.0006196319,0.0006915096,0.0012782357,0.00106405,0.001238697,0.0020451439],"category_scores_gemma":[0.0132912,0.00067547295,0.00088141795,0.00068593014,0.0023293055,0.00176019,0.0011574021,0.0021018572,0.00026421694],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007895296,0.000035834473,0.0007764323,0.000060119663,0.00002531848,0.00007138998,0.00009798063,0.90993994,0.00031188733,0.078020915,0.0002451852,0.010336074],"study_design_scores_gemma":[0.000027510205,0.000021455928,0.00010289695,0.00000970425,0.0000060513103,0.000012417974,0.000009376994,0.92748207,0.00023541263,0.07175067,0.00033456474,0.000007914088],"about_ca_topic_score_codex":0.007624727,"about_ca_topic_score_gemma":0.006490117,"teacher_disagreement_score":0.007624727,"about_ca_system_score_codex":0.0018486847,"about_ca_system_score_gemma":0.0018195119,"threshold_uncertainty_score":0.017086864},"labels":[],"label_agreement":null},{"id":"W2159666783","doi":"10.5555/1838206.1838251","title":"Using spatial hints to improve policy reuse in a reinforcement learning agent","year":2010,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Reinforcement learning; Reuse; Computer science; Exploit; Robustness (evolution); Task (project management); Domain (mathematical analysis); Artificial intelligence; Human–computer interaction; Machine learning; Data science; Computer security; Engineering; Systems engineering","score_opus":0.02532803722935456,"score_gpt":0.3020450763178422,"score_spread":0.27671703908848766,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2159666783","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2705413,0.00043222876,0.72379977,0.0007703025,0.00003849941,0.00015723982,0.000042575393,0.0015618805,0.002656188],"genre_scores_gemma":[0.9114655,0.00009124211,0.08723393,0.00014395801,0.000017979835,0.00007653336,0.00003244269,0.0000492737,0.000889132],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983071,0.0008388276,0.0001101578,0.0002792749,0.00031422064,0.00015043782],"domain_scores_gemma":[0.98824126,0.008061459,0.0011952707,0.0012107368,0.00077652605,0.00051472953],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036506266,0.0013536859,0.0012829144,0.0006400989,0.0004777716,0.0007961596,0.001660962,0.0016570851,0.0013697442],"category_scores_gemma":[0.019319855,0.0005903257,0.00044229985,0.00041833264,0.0014769195,0.0019421062,0.0015862377,0.0014144563,0.0003200843],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008586688,0.0007026162,0.005655595,0.00023331288,0.00019596616,0.0003781275,0.00081443053,0.81748766,0.013724825,0.008893254,0.0008472025,0.15020831],"study_design_scores_gemma":[0.000115915616,0.00029186934,0.00042012776,0.000020177857,0.000045315566,0.00006748961,0.00005054791,0.9852223,0.0056277886,0.007463213,0.0006452534,0.00003002864],"about_ca_topic_score_codex":0.0030708583,"about_ca_topic_score_gemma":0.0033513121,"teacher_disagreement_score":0.0036506266,"about_ca_system_score_codex":0.00075843977,"about_ca_system_score_gemma":0.0015376514,"threshold_uncertainty_score":0.01930654},"labels":[],"label_agreement":null},{"id":"W2159849946","doi":"","title":"Universal Option Models","year":2014,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Function (biology); Consistency (knowledge bases); Domain (mathematical analysis); Task (project management); Relevance (law); Preference; Construct (python library); Mathematical optimization; Computation; Mathematics; Artificial intelligence; Algorithm; Economics","score_opus":0.015026671304967964,"score_gpt":0.20617425160361055,"score_spread":0.1911475802986426,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2159849946","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016850369,0.0005334047,0.9769594,0.0006593491,0.00007160317,0.00008541188,0.00046929118,0.00042040477,0.0039506936],"genre_scores_gemma":[0.68399125,0.0008471473,0.30152503,0.0004773182,0.00017626362,0.00066622527,0.0009924199,0.00020054384,0.011123719],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9971488,0.0011660872,0.00016561072,0.0007657148,0.00043255356,0.00032121988],"domain_scores_gemma":[0.98962015,0.007782902,0.0009292267,0.0006725261,0.00047774366,0.00051735487],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003967432,0.0018911805,0.002269852,0.0014743445,0.0007348809,0.0031129215,0.004109584,0.0031919123,0.011910818],"category_scores_gemma":[0.017961986,0.0012205365,0.0022998224,0.0015926068,0.0024312423,0.006231976,0.0027184335,0.0041591562,0.001051709],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010146662,0.00006778157,0.0010669393,0.00013088886,0.000087822686,0.00016741132,0.00010988236,0.7403853,0.00029176372,0.2383366,0.0014421652,0.017811999],"study_design_scores_gemma":[0.000014758952,0.000027977054,0.00006157592,0.000015321173,0.000010154432,0.00002342659,0.000011019526,0.8641845,0.000093396055,0.13469988,0.00084744964,0.000010575002],"about_ca_topic_score_codex":0.0053149597,"about_ca_topic_score_gemma":0.005568335,"teacher_disagreement_score":0.011910818,"about_ca_system_score_codex":0.0024615454,"about_ca_system_score_gemma":0.0016299745,"threshold_uncertainty_score":0.039845705},"labels":[],"label_agreement":null},{"id":"W2161677453","doi":"10.1109/ical.2008.4636228","title":"Q-learning based multi-robot box-pushing with minimal switching of actions","year":2008,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Robot; Computer science; Reinforcement learning; Action (physics); Robot learning; Artificial intelligence; Mobile robot","score_opus":0.04728716757252466,"score_gpt":0.26318817795017335,"score_spread":0.2159010103776487,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2161677453","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0355826,0.00010014828,0.96243,0.0001191804,0.000041619933,0.00010969648,0.000014506065,0.00049163395,0.0011106357],"genre_scores_gemma":[0.8423811,0.000053130425,0.15556538,0.00010790145,0.000022191909,0.0002251299,0.000035099256,0.00003721609,0.0015728745],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99925977,0.00023876193,0.000047700178,0.0001629363,0.00018927455,0.000101542064],"domain_scores_gemma":[0.9975231,0.0014813456,0.0003010426,0.00017376347,0.0003471949,0.00017344809],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018172546,0.00072527176,0.0013799242,0.00031569836,0.00041715073,0.00054622703,0.0016605224,0.0009883902,0.0017439172],"category_scores_gemma":[0.0036686237,0.00036750527,0.00041674013,0.00028331717,0.0011561557,0.000784281,0.0010651022,0.0009989475,0.00026384814],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002852529,0.00021044849,0.0008375419,0.00011596118,0.00006744696,0.00016505379,0.00012551864,0.87525433,0.006958076,0.01028236,0.00065408007,0.105043955],"study_design_scores_gemma":[0.000034338827,0.0000837719,0.000080671445,0.0000032324754,0.000004572859,0.000015622847,0.0000026877608,0.9972451,0.0008426091,0.0015221874,0.00015977456,0.0000056101794],"about_ca_topic_score_codex":0.0025972393,"about_ca_topic_score_gemma":0.0012794787,"teacher_disagreement_score":0.0025972393,"about_ca_system_score_codex":0.0005398482,"about_ca_system_score_gemma":0.0011950345,"threshold_uncertainty_score":0.0096107125},"labels":[],"label_agreement":null},{"id":"W2161717678","doi":"","title":"MDPs with Non-Deterministic Policies.","year":2009,"lang":"en","type":"article","venue":"PubMed","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Markov decision process; Flexibility (engineering); Context (archaeology); Mathematical optimization; Optimal decision; Integer (computer science); Markov process; Operations research; Artificial intelligence; Decision tree; Mathematics","score_opus":0.014300349578223721,"score_gpt":0.21057131257444645,"score_spread":0.19627096299622274,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2161717678","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014503048,0.0007989463,0.9760562,0.0009202,0.000100898316,0.00020003604,0.0002963056,0.00022989056,0.006894412],"genre_scores_gemma":[0.62442124,0.0012856463,0.36306813,0.00046850747,0.000114986266,0.0011516422,0.0005164633,0.00007327425,0.008900127],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9984139,0.000811486,0.00010637956,0.00025082604,0.0002858732,0.00013143341],"domain_scores_gemma":[0.9929931,0.005591616,0.00069031504,0.00025854568,0.00026582967,0.00020049598],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002361781,0.0013161923,0.0012853465,0.00064248004,0.0005617105,0.001433163,0.0014420575,0.0019426357,0.0049878825],"category_scores_gemma":[0.01036349,0.00072897685,0.0011679461,0.00077011227,0.0017497286,0.0019212726,0.001619547,0.0022363383,0.00058148947],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007197137,0.000047019148,0.00044904495,0.00011622656,0.000040801693,0.00010070625,0.00005232984,0.8981551,0.0002769154,0.08935973,0.00075218803,0.010577867],"study_design_scores_gemma":[0.00003405883,0.00003491809,0.00007066307,0.000020090016,0.000014045999,0.000022861808,0.000015114524,0.94200855,0.000261138,0.055905793,0.0016043094,0.000008441938],"about_ca_topic_score_codex":0.0040774294,"about_ca_topic_score_gemma":0.0043893345,"teacher_disagreement_score":0.0049878825,"about_ca_system_score_codex":0.0015907859,"about_ca_system_score_gemma":0.002190293,"threshold_uncertainty_score":0.016686201},"labels":[],"label_agreement":null},{"id":"W2162664081","doi":"","title":"Dual Temporal Difference Learning","year":2009,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Dual (grammatical number); Convergence (economics); Temporal difference learning; Reinforcement learning; Computer science; Artificial intelligence; Dynamic programming; Basis (linear algebra); Machine learning; Algorithm; Mathematical optimization; Mathematics","score_opus":0.016511148029155924,"score_gpt":0.2444750998830643,"score_spread":0.22796395185390836,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2162664081","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018830465,0.00019573692,0.97655404,0.00036064486,0.00007434786,0.000037513095,0.000040168492,0.00011799906,0.0037890843],"genre_scores_gemma":[0.79876316,0.00028031506,0.1944079,0.00031326624,0.00006950675,0.00016979137,0.00012555537,0.00006339196,0.0058071106],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99893826,0.00043747236,0.000052956853,0.00021709532,0.00026234236,0.0000918605],"domain_scores_gemma":[0.9969438,0.0017176026,0.0002492593,0.00035432854,0.0005283028,0.00020676602],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025329406,0.00055662205,0.000754662,0.00047329746,0.00033016404,0.0010891634,0.0013931454,0.0010068187,0.0044662086],"category_scores_gemma":[0.010892136,0.0003056839,0.00044270017,0.0004944818,0.0015575843,0.002564791,0.0021060659,0.0020677862,0.00042897777],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040162235,0.00023509131,0.0014962051,0.00018293099,0.00007764975,0.00007739572,0.00016391533,0.2884901,0.0048785387,0.5337355,0.002237102,0.168024],"study_design_scores_gemma":[0.000026139543,0.000071150054,0.00008191165,0.000011059032,0.000008061853,0.0000275907,0.000009547556,0.88846374,0.0011236471,0.109126054,0.0010434411,0.000007655803],"about_ca_topic_score_codex":0.00079307245,"about_ca_topic_score_gemma":0.00064547604,"teacher_disagreement_score":0.0044662086,"about_ca_system_score_codex":0.001037061,"about_ca_system_score_gemma":0.0009954362,"threshold_uncertainty_score":0.0149409175},"labels":[],"label_agreement":null},{"id":"W2162926979","doi":"10.1109/cdc.2009.5400662","title":"Arbitrarily modulated Markov decision processes","year":2009,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Markov decision process; Transition (genetics); Computer science; Markov chain; Markov process; Markov model; Partially observable Markov decision process; Decision theory; Artificial intelligence; Mathematical optimization; Machine learning; Algorithm; Mathematics; Statistics","score_opus":0.009695192704821434,"score_gpt":0.24127695218328302,"score_spread":0.23158175947846157,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2162926979","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039558742,0.0005403887,0.9526248,0.0008961754,0.00017610972,0.00009816838,0.00014817683,0.00019349398,0.005764042],"genre_scores_gemma":[0.90954757,0.0007236749,0.082269974,0.00038705932,0.00022796888,0.00023829722,0.00016517908,0.000031972057,0.0064082234],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99830174,0.0006949944,0.00006606569,0.0003965015,0.00030560174,0.00023508946],"domain_scores_gemma":[0.9956416,0.0028839486,0.0006225884,0.00029750823,0.00025841896,0.00029586855],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017671258,0.0011519903,0.0013466887,0.00051710533,0.0006203085,0.001662534,0.001846654,0.0021635701,0.0034731338],"category_scores_gemma":[0.009189716,0.0004896552,0.0009899989,0.0008765295,0.0021567836,0.0018611283,0.0018778695,0.0027977498,0.00071798527],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017155682,0.000090797585,0.0012466495,0.000100525715,0.000069655376,0.00040675406,0.00015370177,0.59175664,0.0021142452,0.38765642,0.0010811543,0.015151879],"study_design_scores_gemma":[0.000028269213,0.00003364796,0.00010228383,0.000007815595,0.000012544829,0.000026321622,0.000007028614,0.91267556,0.00025065313,0.08611479,0.0007310506,0.00001008063],"about_ca_topic_score_codex":0.0026350503,"about_ca_topic_score_gemma":0.0016356652,"teacher_disagreement_score":0.0034731338,"about_ca_system_score_codex":0.0013126051,"about_ca_system_score_gemma":0.0009657545,"threshold_uncertainty_score":0.011618793},"labels":[],"label_agreement":null},{"id":"W2163126463","doi":"","title":"Online Discovery and Learning of Predictive State Representations","year":2005,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":50,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Machine learning; Outcome (game theory); Artificial intelligence; Gradient descent; State (computer science); Current (fluid); Algorithm; Monte Carlo method; Data mining; Artificial neural network; Mathematics","score_opus":0.013972898074078937,"score_gpt":0.27080232391216796,"score_spread":0.25682942583808904,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2163126463","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03292457,0.00017978466,0.96416736,0.00043341724,0.000028009403,0.00007524841,0.00015190976,0.0010370075,0.0010027143],"genre_scores_gemma":[0.7637321,0.00019202086,0.23309404,0.00023368634,0.0000546264,0.00030264037,0.0007161121,0.00013777349,0.0015370474],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982046,0.00064562325,0.0001242508,0.00044192417,0.0004049697,0.00017868534],"domain_scores_gemma":[0.98376644,0.01219103,0.0011828033,0.0013181742,0.0012336487,0.00030787016],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003230471,0.0011421965,0.001893662,0.0013748091,0.0006020138,0.0013312142,0.0029226537,0.0018478639,0.0018755114],"category_scores_gemma":[0.021222347,0.00092199433,0.0007481032,0.0010388523,0.0018833552,0.0036413493,0.0019650308,0.002964907,0.00044710358],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020258047,0.00017505865,0.003066506,0.00012863235,0.000072802424,0.00013964815,0.00013681447,0.8506566,0.0011533488,0.027823158,0.0026494917,0.11379528],"study_design_scores_gemma":[0.000009710092,0.000012995943,0.000092716,0.0000047652907,0.000003892337,0.000009655426,0.0000046184223,0.98985356,0.00029854715,0.009588757,0.0001167257,0.0000040346727],"about_ca_topic_score_codex":0.0052356045,"about_ca_topic_score_gemma":0.0043913648,"teacher_disagreement_score":0.0052356045,"about_ca_system_score_codex":0.0016045346,"about_ca_system_score_gemma":0.0022827066,"threshold_uncertainty_score":0.017084539},"labels":[],"label_agreement":null},{"id":"W2164065344","doi":"10.1109/iros.2007.4399526","title":"Decision theoretic task coordination for a visually-guided interactive mobile robot","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Partially observable Markov decision process; Heuristics; Computer science; Mobile robot; Robot; Markov decision process; Task (project management); Artificial intelligence; Social robot; Human–computer interaction; Planner; Process (computing); Markov process; Robot learning; Behavior-based robotics; Robot control; Machine learning; Markov chain; Markov model; Engineering","score_opus":0.014687924549861111,"score_gpt":0.3253183112698323,"score_spread":0.3106303867199712,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2164065344","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06524386,0.00011140833,0.9295893,0.0003005946,0.000023078168,0.00007908265,0.00003899512,0.00037185443,0.0042418963],"genre_scores_gemma":[0.8677311,0.00006903885,0.12998559,0.000050360937,0.0000105809995,0.00015698782,0.000047535716,0.000031344636,0.0019175321],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99968004,0.000105670246,0.000012684001,0.00008271387,0.00007123048,0.000047710506],"domain_scores_gemma":[0.99954885,0.00022975185,0.00008663433,0.00003085745,0.000046268076,0.00005764391],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005647649,0.0005575641,0.00045413774,0.00022884006,0.00046150797,0.0005772461,0.0007437203,0.00078317604,0.001787467],"category_scores_gemma":[0.0015522379,0.00029188776,0.00036188748,0.00017403264,0.0010562284,0.00064290175,0.0009792821,0.0007342373,0.0002559703],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000110891146,0.00006341743,0.00036964368,0.000051928844,0.00002018461,0.00013380134,0.000115229726,0.96314335,0.0043079276,0.019301137,0.00032963455,0.012052881],"study_design_scores_gemma":[0.000023752165,0.000042113188,0.00007876127,0.0000031443153,0.000004324931,0.00001284692,0.00001568565,0.9920665,0.00061683805,0.0068191397,0.00031224775,0.000004628271],"about_ca_topic_score_codex":0.0064532366,"about_ca_topic_score_gemma":0.0050792173,"teacher_disagreement_score":0.0064532366,"about_ca_system_score_codex":0.0011048933,"about_ca_system_score_gemma":0.0015520169,"threshold_uncertainty_score":0.01283139},"labels":[],"label_agreement":null},{"id":"W2164327805","doi":"10.1115/1.2764516","title":"Decentralized Coordinated Motion Control of Two Hydraulic Actuators Handling a Common Object","year":2007,"lang":"en","type":"article","venue":"Journal of Dynamic Systems Measurement and Control","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Actuator; Trajectory; Computer science; Task (project management); Object (grammar); Control theory (sociology); Position (finance); Artificial intelligence; Control engineering; Control (management); Engineering","score_opus":0.013405342807333475,"score_gpt":0.2483025158322252,"score_spread":0.23489717302489174,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2164327805","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11718799,0.000096942764,0.8785095,0.00023575463,0.00006388324,0.00008044973,0.000014541683,0.00039287438,0.0034180249],"genre_scores_gemma":[0.96920836,0.00002967136,0.02925497,0.000026577305,0.000011694801,0.00006604851,0.000010130953,0.000008621619,0.0013838916],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996737,0.000078200515,0.000014685794,0.000092303155,0.000088998146,0.00005210987],"domain_scores_gemma":[0.99923015,0.0002278669,0.0002314917,0.000102593505,0.0001270854,0.000080839425],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00076225644,0.00045708756,0.0004548389,0.0001789998,0.00040273665,0.00041918154,0.0008187961,0.0004959681,0.0010273287],"category_scores_gemma":[0.0013794295,0.00024081547,0.00024977466,0.00014726541,0.00096961844,0.00050628016,0.00087397284,0.00047886567,0.00016780067],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027851653,0.00014186242,0.00091245875,0.00006812374,0.00003579845,0.00027359748,0.00017859649,0.897693,0.034825593,0.010544879,0.0004429518,0.054604646],"study_design_scores_gemma":[0.00004434738,0.0001650174,0.0002800853,0.0000031063728,0.0000058490027,0.000029829556,0.000013741179,0.9935633,0.002888504,0.0023624178,0.00063559756,0.000008224183],"about_ca_topic_score_codex":0.002451186,"about_ca_topic_score_gemma":0.0017891455,"teacher_disagreement_score":0.002451186,"about_ca_system_score_codex":0.0005297483,"about_ca_system_score_gemma":0.0008505612,"threshold_uncertainty_score":0.004873812},"labels":[],"label_agreement":null},{"id":"W2164750193","doi":"10.5430/air.v1n2p1","title":"Combining coordination mechanisms to improve performance in multi-robot teams","year":2012,"lang":"en","type":"article","venue":"Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Stigmergy; Negotiation; Robot; Artificial intelligence; Domain (mathematical analysis); Human–computer interaction; Mechanism (biology); Multi-agent system; Distributed computing","score_opus":0.17958176777957213,"score_gpt":0.4121311132034148,"score_spread":0.23254934542384265,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2164750193","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3612828,0.00067291904,0.62930876,0.00047140793,0.0001001007,0.00016188074,0.00002151351,0.0014995543,0.0064810775],"genre_scores_gemma":[0.9412794,0.000087583045,0.057736997,0.0000458164,0.000025561245,0.00007382804,0.000018209841,0.000040379575,0.0006920525],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99796104,0.0008065891,0.00017901318,0.00029136598,0.0005273245,0.00023454911],"domain_scores_gemma":[0.993989,0.00312345,0.00087734015,0.0011729302,0.00050453166,0.00033278027],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003860863,0.0010766622,0.00092118216,0.000736848,0.0005424395,0.001303266,0.001740238,0.0009829904,0.0012802606],"category_scores_gemma":[0.008826022,0.00048010793,0.00036680297,0.00042091613,0.00097059726,0.002361435,0.0035436484,0.0012399713,0.00032831435],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005312117,0.00064985594,0.0057525174,0.00022684039,0.0001820044,0.0001521236,0.0003231876,0.73668754,0.027164819,0.014693498,0.0008907464,0.21274562],"study_design_scores_gemma":[0.00014571901,0.00054384244,0.001196737,0.000021306918,0.00005219585,0.00006415022,0.000059482274,0.9789217,0.009986833,0.007886531,0.0010882727,0.000033209864],"about_ca_topic_score_codex":0.0006634235,"about_ca_topic_score_gemma":0.00084377796,"teacher_disagreement_score":0.003860863,"about_ca_system_score_codex":0.0005993796,"about_ca_system_score_gemma":0.00066505076,"threshold_uncertainty_score":0.020418406},"labels":[],"label_agreement":null},{"id":"W2164941694","doi":"10.48550/arxiv.1206.6444","title":"Statistical Linear Estimation with Penalized Estimators: an Application to Reinforcement Learning","year":2012,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Estimator; Reinforcement learning; Regularization (linguistics); Applied mathematics; Mathematics; Mathematical optimization; Computer science; Feature (linguistics); Inverse problem; Function (biology); Algorithm; Artificial intelligence; Statistics","score_opus":0.0385205685199137,"score_gpt":0.21785937348316717,"score_spread":0.17933880496325347,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2164941694","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0036760517,0.00025669864,0.9949508,0.00030095692,0.000020862617,0.00001631518,0.000007288935,0.000050211704,0.0007207969],"genre_scores_gemma":[0.6428519,0.000978329,0.35202193,0.0004150759,0.00027087974,0.0003420507,0.00006909657,0.00013854765,0.0029121554],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99663395,0.0022390743,0.000115204195,0.00033989357,0.0005394049,0.00013241383],"domain_scores_gemma":[0.97699136,0.019513538,0.0012394863,0.00073073007,0.0011791098,0.00034584678],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0065967957,0.0012545544,0.0015442175,0.00078235735,0.00038598743,0.0014170834,0.0016080898,0.0022185862,0.0013807946],"category_scores_gemma":[0.028513413,0.00059900177,0.000815275,0.00083656015,0.0033764986,0.0018254877,0.0027172163,0.0027757462,0.00019267727],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000057424164,0.00005054647,0.00082311366,0.000129257,0.000078564866,0.00011406553,0.00010876665,0.861536,0.0011281703,0.11657588,0.0006125714,0.018785644],"study_design_scores_gemma":[0.000006563948,0.00001867001,0.00003851237,0.000008897552,0.000004219026,0.000008976055,0.0000033985768,0.9748024,0.000141598,0.024809746,0.00015185466,0.000005197182],"about_ca_topic_score_codex":0.0021825742,"about_ca_topic_score_gemma":0.0014596769,"teacher_disagreement_score":0.0065967957,"about_ca_system_score_codex":0.0015896616,"about_ca_system_score_gemma":0.0011573469,"threshold_uncertainty_score":0.034887552},"labels":[],"label_agreement":null},{"id":"W2165044515","doi":"","title":"Symbolic Dynamic Programming for Continuous State and Observation POMDPs","year":2012,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Markov decision process; Dynamic programming; Computer science; Observable; Mathematical optimization; State (computer science); Partially observable Markov decision process; Set (abstract data type); Discrete time and continuous time; Theoretical computer science; Markov process; Algorithm; Mathematics","score_opus":0.020659966466638514,"score_gpt":0.2658637831583126,"score_spread":0.2452038166916741,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2165044515","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005185214,0.00012397497,0.99183685,0.000118750104,0.000012142615,0.000021621587,0.000044190474,0.00010847136,0.0025487149],"genre_scores_gemma":[0.5744158,0.00046312457,0.41863322,0.0000919731,0.00004984032,0.00048653258,0.00028503148,0.00014540556,0.005429082],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988329,0.00044387675,0.000056872163,0.00017619527,0.0003748061,0.000115337316],"domain_scores_gemma":[0.99722606,0.0022439177,0.00018264046,0.000101263424,0.00016009655,0.00008602792],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017303523,0.0010370165,0.0013604118,0.0006584056,0.00056173676,0.0016790678,0.0014108395,0.0012395716,0.0033263713],"category_scores_gemma":[0.006166259,0.00067796913,0.0011096881,0.0010837606,0.0023647747,0.0018131445,0.0020494636,0.0025797873,0.00034515152],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000020096928,0.00001384693,0.00014319347,0.000049735863,0.000014355969,0.00004051128,0.00007610265,0.8986948,0.00024680412,0.09127783,0.00019083223,0.009231856],"study_design_scores_gemma":[0.000004990455,0.0000041997005,0.00001442002,0.000004816358,0.0000017896741,0.000003650705,0.0000069167963,0.96958053,0.00007712937,0.03005527,0.00024367598,0.0000025667723],"about_ca_topic_score_codex":0.0062202844,"about_ca_topic_score_gemma":0.0055860644,"teacher_disagreement_score":0.0062202844,"about_ca_system_score_codex":0.0020021992,"about_ca_system_score_gemma":0.0021559566,"threshold_uncertainty_score":0.014527023},"labels":[],"label_agreement":null},{"id":"W2165235114","doi":"","title":"Using Free Energies to Represent Q-values in a Multiagent Reinforcement Learning Task","year":2000,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Q-learning; Markov decision process; Computer science; Task (project management); Artificial intelligence; Markov chain; Sampling (signal processing); Markov process; Machine learning; State (computer science); Representation (politics); Gibbs sampling; Table (database); Action (physics); Algorithm; Data mining; Mathematics; Statistics; Engineering","score_opus":0.035425327012383365,"score_gpt":0.293271296584887,"score_spread":0.2578459695725036,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2165235114","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033083573,0.00016103845,0.96524596,0.0002627195,0.000023180284,0.000043410055,0.00003881297,0.00019637153,0.0009448888],"genre_scores_gemma":[0.8103815,0.00017136363,0.18722342,0.00011968929,0.00003461444,0.0002465653,0.00010525648,0.0000981615,0.0016195173],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99890816,0.00067990064,0.00004954195,0.00016373741,0.00010868749,0.00008999063],"domain_scores_gemma":[0.9918533,0.0067864945,0.00044675806,0.00033765854,0.0003694894,0.00020641104],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003969552,0.0009745232,0.001477511,0.00082104944,0.0005449431,0.0013769515,0.0018365033,0.0019377465,0.0022577613],"category_scores_gemma":[0.013255788,0.0008032052,0.000606782,0.0006970788,0.0019701507,0.0027244785,0.001390556,0.0017778339,0.00030695132],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000035843626,0.0000169928,0.00032019184,0.000014833631,0.000014046676,0.000022100907,0.000029132802,0.98442763,0.00015328522,0.00764507,0.0001417178,0.007179136],"study_design_scores_gemma":[0.000005381834,0.000009964012,0.000035864887,0.0000026607504,0.0000018625603,0.000003814967,0.0000024961705,0.9913244,0.000066664696,0.008485475,0.000058100977,0.000003266963],"about_ca_topic_score_codex":0.0059371223,"about_ca_topic_score_gemma":0.0036553266,"teacher_disagreement_score":0.0059371223,"about_ca_system_score_codex":0.001753666,"about_ca_system_score_gemma":0.001235997,"threshold_uncertainty_score":0.020993233},"labels":[],"label_agreement":null},{"id":"W2167748058","doi":"","title":"Average Reward Optimization Objective In Partially Observable Domains","year":2013,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Observability; Dimension (graph theory); Observable; Representation (politics); Function (biology); Mathematics; Process (computing); Mathematical optimization; Bellman equation; Computer science; Computation; Key (lock); Applied mathematics; Algorithm","score_opus":0.012999340964716806,"score_gpt":0.22070248817275684,"score_spread":0.20770314720804003,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2167748058","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.045994304,0.00045268703,0.9495853,0.00059264264,0.000026478898,0.000039530914,0.0001424197,0.00019580507,0.0029709295],"genre_scores_gemma":[0.90643996,0.00041003854,0.089793,0.00009576825,0.00004277743,0.0001211473,0.00015241063,0.0000767193,0.002868109],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991165,0.00038587258,0.000038648373,0.00018689712,0.0001612725,0.0001109007],"domain_scores_gemma":[0.9969836,0.0021940833,0.00034238686,0.00013214964,0.00020800093,0.00013979552],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001910174,0.0009782667,0.0016116281,0.000572169,0.0003940454,0.0013652205,0.0010464041,0.0013609676,0.0017942274],"category_scores_gemma":[0.006668587,0.00046665475,0.000664202,0.00060712156,0.0015925878,0.0020876862,0.0012449138,0.0014325946,0.00018099272],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000029795734,0.000015136363,0.0002130818,0.000042630545,0.000016097587,0.000037568047,0.000027343614,0.95847243,0.00032030928,0.03585707,0.00025828503,0.0047102785],"study_design_scores_gemma":[0.0000047212334,0.00001249416,0.000050772807,0.0000049198657,0.0000029597204,0.0000044210456,0.0000037452571,0.98041487,0.00011719321,0.019280808,0.000099569625,0.0000034402483],"about_ca_topic_score_codex":0.0047992445,"about_ca_topic_score_gemma":0.002273174,"teacher_disagreement_score":0.0047992445,"about_ca_system_score_codex":0.0018514461,"about_ca_system_score_gemma":0.0011837751,"threshold_uncertainty_score":0.013433218},"labels":[],"label_agreement":null},{"id":"W2169183587","doi":"10.1109/ijcnn.2006.246689","title":"Opposition-Based Q(&amp;amp;#955;) Algorithm","year":2006,"lang":"en","type":"article","venue":"The 2006 IEEE International Joint Conference on Neural Network Proceedings","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Opposition (politics); Reinforcement learning; Computer science; Lambda; Algorithm; Artificial intelligence; Law; Political science; Physics","score_opus":0.053300843179456,"score_gpt":0.27574545304177817,"score_spread":0.22244460986232217,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2169183587","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008835741,0.00009012158,0.98619854,0.00018092057,0.000055738437,0.000096699165,0.000034366814,0.00047354135,0.004034367],"genre_scores_gemma":[0.4422873,0.00012386027,0.5466424,0.000462459,0.000072132294,0.0004906067,0.00016531539,0.00015078977,0.009605201],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99916685,0.00032406728,0.000043281107,0.00011760447,0.00023554555,0.00011267505],"domain_scores_gemma":[0.99840647,0.00092618604,0.00008026338,0.00011564012,0.00035658616,0.00011494435],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019188346,0.0005443676,0.00086760067,0.0005875477,0.00048934517,0.00067253516,0.0020120447,0.0013265995,0.009186946],"category_scores_gemma":[0.004323493,0.00025733965,0.0003865298,0.0005563585,0.00096764497,0.0010729117,0.0016893672,0.0012821149,0.0014365757],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005996911,0.0003515131,0.001153429,0.0002060871,0.00007919999,0.00015139714,0.0001964984,0.29387143,0.004724749,0.09735903,0.009455307,0.59185165],"study_design_scores_gemma":[0.0001236991,0.000176145,0.00013950995,0.000010092276,0.000014055943,0.000063190215,0.0000171543,0.9689229,0.0015045668,0.026039926,0.0029774888,0.0000112569005],"about_ca_topic_score_codex":0.0014927818,"about_ca_topic_score_gemma":0.001450769,"teacher_disagreement_score":0.009186946,"about_ca_system_score_codex":0.00068628753,"about_ca_system_score_gemma":0.0013588293,"threshold_uncertainty_score":0.030733407},"labels":[],"label_agreement":null},{"id":"W2169190416","doi":"","title":"Using bisimulation for policy transfer in MDPs (Extended Abstract)","year":2010,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Markov decision process; Bellman equation; Set (abstract data type); Function (biology); Action (physics); Value (mathematics); Computer science; State (computer science); Mathematical economics; Markov process; Artificial intelligence; Mathematics; Algorithm; Machine learning; Statistics","score_opus":0.048726002218220925,"score_gpt":0.3417722960033684,"score_spread":0.29304629378514746,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2169190416","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0038295959,0.000503979,0.9896841,0.0003746001,0.00006759519,0.000097291406,0.00013346158,0.00033069818,0.004978672],"genre_scores_gemma":[0.46569666,0.0024038453,0.5149169,0.0009148582,0.00032466956,0.0018563798,0.001022764,0.0008685903,0.011995295],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99641776,0.0017004805,0.00020226999,0.0006821735,0.00068388507,0.00031348006],"domain_scores_gemma":[0.9856873,0.011380809,0.00089501415,0.00069955835,0.0009166238,0.000420633],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0055784574,0.0028205265,0.002648564,0.001742126,0.00092170667,0.0026409784,0.0027089603,0.002637893,0.014184264],"category_scores_gemma":[0.026991677,0.0012475882,0.003007824,0.0020231928,0.0025579846,0.004383248,0.005154373,0.005178037,0.0023677994],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000081919636,0.00009341439,0.00032593912,0.00022832537,0.00006393338,0.00008279735,0.00009776448,0.823595,0.0004241361,0.15013254,0.0011213644,0.02375291],"study_design_scores_gemma":[0.000016659578,0.000033554603,0.000028082935,0.000038820886,0.000011583394,0.000015146891,0.000008253022,0.914894,0.00019644735,0.08361085,0.0011350727,0.000011595486],"about_ca_topic_score_codex":0.0068824445,"about_ca_topic_score_gemma":0.0034364525,"teacher_disagreement_score":0.014184264,"about_ca_system_score_codex":0.0037484232,"about_ca_system_score_gemma":0.0036743153,"threshold_uncertainty_score":0.04745108},"labels":[],"label_agreement":null},{"id":"W2169375932","doi":"10.5555/2447556.2447651","title":"Integrating a robot in a tabletop reservoir engineering application","year":2013,"lang":"en","type":"article","venue":"OCAD University Open Research Repository (OCAD University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Visualization; Human–computer interaction; Simple (philosophy); Robot; Data visualization; Artificial intelligence","score_opus":0.03200415483870318,"score_gpt":0.2613764750102883,"score_spread":0.22937232017158515,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2169375932","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19444431,0.00027760555,0.77519125,0.00044199164,0.00009541417,0.0009032465,0.00030340816,0.015597248,0.012745613],"genre_scores_gemma":[0.398963,0.0002849005,0.58645225,0.00024188648,0.000032276024,0.0003755895,0.00024692324,0.00044797384,0.012955224],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.999495,0.0000917727,0.000025940786,0.00014028887,0.00018916643,0.0000576601],"domain_scores_gemma":[0.998922,0.00048008456,0.00008313543,0.00021127622,0.00011785384,0.00018563622],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005170846,0.0008085234,0.0005749215,0.00035565352,0.0004794766,0.0012271216,0.0021632132,0.0010950853,0.012477043],"category_scores_gemma":[0.0018028558,0.0005588452,0.0005069026,0.0002338739,0.0006553286,0.0014112924,0.002013147,0.0006991986,0.0024703883],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016863772,0.001239446,0.006953527,0.001498483,0.00016223406,0.0040304386,0.0035297489,0.03635576,0.49069318,0.0055238865,0.010778106,0.4375489],"study_design_scores_gemma":[0.00050794333,0.0057052854,0.017536666,0.0003738846,0.00027981368,0.0064224387,0.0018358737,0.5352393,0.21330842,0.0044639455,0.2138394,0.00048707085],"about_ca_topic_score_codex":0.0015276509,"about_ca_topic_score_gemma":0.0021658796,"teacher_disagreement_score":0.012477043,"about_ca_system_score_codex":0.00019645144,"about_ca_system_score_gemma":0.0007205465,"threshold_uncertainty_score":0.04173988},"labels":[],"label_agreement":null},{"id":"W2169619645","doi":"10.5555/2484920.2485084","title":"Smart exploration in reinforcement learning using absolute temporal difference errors","year":2013,"lang":"en","type":"article","venue":"Adaptive Agents and Multi-Agents Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Temporal difference learning; Computer science; State (computer science); Function (biology); Artificial intelligence; Function approximation; Control (management); Machine learning; Algorithm; Artificial neural network","score_opus":0.10786902931368043,"score_gpt":0.29261669850862043,"score_spread":0.18474766919494,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2169619645","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02723811,0.00022446939,0.971386,0.000106082705,0.000027116035,0.000022733411,0.0000103711955,0.00015445604,0.00083053164],"genre_scores_gemma":[0.88897383,0.00016865991,0.10920404,0.000059496804,0.00003397859,0.00010964135,0.00003137017,0.00006312833,0.0013558455],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99903464,0.00041345137,0.00005918626,0.00015251481,0.00026848225,0.00007181858],"domain_scores_gemma":[0.9947207,0.0040114485,0.00043171042,0.00028622895,0.00034554885,0.00020434178],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025821356,0.0008085276,0.0010390744,0.00048601048,0.0002595262,0.00085009774,0.0011260418,0.0008664833,0.0010680992],"category_scores_gemma":[0.010510602,0.0004126291,0.00039861017,0.00041096,0.0016786524,0.002003563,0.0015359757,0.0014193375,0.00014367164],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015743973,0.00004524558,0.0007917265,0.00007360104,0.00003533576,0.000052790252,0.00007621475,0.923762,0.0018596351,0.036724716,0.00023850943,0.03618286],"study_design_scores_gemma":[0.00001076811,0.00002396163,0.000041817333,0.0000034761258,0.0000022146337,0.0000058332084,0.0000018289896,0.99226475,0.0003455589,0.0072167446,0.00007999109,0.0000030912083],"about_ca_topic_score_codex":0.0019112634,"about_ca_topic_score_gemma":0.001289838,"teacher_disagreement_score":0.0025821356,"about_ca_system_score_codex":0.00084907736,"about_ca_system_score_gemma":0.00081730296,"threshold_uncertainty_score":0.013655782},"labels":[],"label_agreement":null},{"id":"W2172123610","doi":"10.1142/s0217979200002077","title":"SELF-ADAPTING REACTIVE AUTONOMOUS AGENTS","year":2000,"lang":"en","type":"article","venue":"International Journal of Modern Physics B","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Computer science; Reinforcement learning; Autonomous agent; Adaptation (eye); Architecture; Distributed computing; Autonomous system (mathematics); Multi-agent system; Artificial intelligence; Human–computer interaction","score_opus":0.019786536530311302,"score_gpt":0.27127720933122484,"score_spread":0.25149067280091353,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2172123610","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009834029,0.00031547164,0.9810527,0.00014472821,0.00013164504,0.00007198619,0.000011603378,0.00084547704,0.007592368],"genre_scores_gemma":[0.47199997,0.00080173986,0.50952536,0.0004346535,0.00013951428,0.00036965372,0.000106615,0.00022609714,0.016396448],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99958426,0.00007378326,0.0000230152,0.00007983794,0.0002077804,0.00003126776],"domain_scores_gemma":[0.9995271,0.0001622485,0.00005341191,0.00008372859,0.00013686299,0.000036666184],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00042797072,0.00041377736,0.00043755939,0.00029864244,0.00033562392,0.00064529304,0.0012227364,0.0006219234,0.0015539082],"category_scores_gemma":[0.0013469439,0.00020088423,0.0003436186,0.00017640674,0.00063518155,0.00077937497,0.00077107607,0.00079891185,0.00048514493],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008840188,0.0001802628,0.0010324139,0.00021227846,0.00010574721,0.00033541973,0.000334009,0.4440263,0.043913588,0.18119949,0.0055844677,0.3229877],"study_design_scores_gemma":[0.000025376477,0.000054219567,0.00013734442,0.00001626666,0.000019839132,0.00010617134,0.000019948224,0.9562218,0.0065614884,0.016017137,0.020801622,0.000018712439],"about_ca_topic_score_codex":0.000892827,"about_ca_topic_score_gemma":0.00082830404,"teacher_disagreement_score":0.0015539082,"about_ca_system_score_codex":0.00031843732,"about_ca_system_score_gemma":0.0004103915,"threshold_uncertainty_score":0.005198419},"labels":[],"label_agreement":null},{"id":"W2174786457","doi":"10.48550/arxiv.1511.06342","title":"Actor-Mimic: Deep Multitask and Transfer Reinforcement Learning","year":2015,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":208,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Exploit; Artificial intelligence; Transfer of learning; Set (abstract data type); Task (project management); Machine learning; Engineering","score_opus":0.07661995159032055,"score_gpt":0.19594166404524566,"score_spread":0.11932171245492511,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2174786457","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01608139,0.00018721719,0.97867227,0.00031835298,0.00006556316,0.00005294691,0.00004377806,0.0009716284,0.003606827],"genre_scores_gemma":[0.78593785,0.00019209117,0.20665702,0.00022541052,0.000052118907,0.0002368795,0.00012290716,0.00014937906,0.006426381],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99950564,0.00021268982,0.000017836837,0.00009817468,0.000116982694,0.000048618764],"domain_scores_gemma":[0.99915755,0.00041738377,0.0000987707,0.00015719372,0.000087353415,0.0000818663],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012719302,0.0007304588,0.00063432916,0.00025451867,0.00027189162,0.00062061066,0.0017667734,0.0010139013,0.0026221692],"category_scores_gemma":[0.0036523426,0.0003778806,0.0004145399,0.0002616442,0.0011097937,0.001295918,0.0013821793,0.0018037277,0.00050130783],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012253229,0.00011543965,0.00079523044,0.00007630991,0.00006381912,0.00010005211,0.00007857683,0.87031645,0.0030326732,0.05194716,0.0030023463,0.07034942],"study_design_scores_gemma":[0.0000074216623,0.000019251542,0.000029419047,0.0000026402051,0.0000022380693,0.000008883296,0.0000019989657,0.9875773,0.00047834372,0.0113284495,0.0005415728,0.000002564677],"about_ca_topic_score_codex":0.0021327853,"about_ca_topic_score_gemma":0.002278087,"teacher_disagreement_score":0.0026221692,"about_ca_system_score_codex":0.0008899579,"about_ca_system_score_gemma":0.0010617155,"threshold_uncertainty_score":0.008772016},"labels":[],"label_agreement":null},{"id":"W2181804636","doi":"","title":"Autonomous Navigation Through Crowds","year":2011,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Crowds; Computer science; Presentation (obstetrics); Artificial intelligence; Reinforcement learning; Computer vision; Mobile robot; Human–computer interaction; Robot; Computer security","score_opus":0.04958624137517291,"score_gpt":0.254975453139823,"score_spread":0.20538921176465008,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2181804636","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05786562,0.00075955526,0.92626023,0.00078902656,0.00015178377,0.00010441849,0.00017842946,0.0012027634,0.01268813],"genre_scores_gemma":[0.9185066,0.00046441867,0.07429884,0.0001654938,0.00006519929,0.00013238406,0.00019040373,0.00010324141,0.0060734763],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99935824,0.00023249291,0.000020589234,0.0001518166,0.0001757895,0.000061030714],"domain_scores_gemma":[0.99911886,0.00047054602,0.00008219294,0.00014360742,0.000114936454,0.00006989569],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00065883057,0.0005801506,0.0007278545,0.00043284436,0.0006950821,0.0008331557,0.0009084447,0.0007619845,0.0017659814],"category_scores_gemma":[0.0032152706,0.00034945057,0.00047746362,0.00041866596,0.0012672498,0.0012995068,0.002566105,0.0007729454,0.0004904353],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018581901,0.000041034345,0.0012324365,0.00017486102,0.00007550503,0.00027325528,0.00048874074,0.8562267,0.005070619,0.05939485,0.003502932,0.07333327],"study_design_scores_gemma":[0.000028031343,0.000035488814,0.00030877622,0.000022009799,0.0000083481955,0.00006237415,0.0001094618,0.90039206,0.0012692778,0.09168442,0.0060583176,0.000021439193],"about_ca_topic_score_codex":0.0059511876,"about_ca_topic_score_gemma":0.0035311836,"teacher_disagreement_score":0.0059511876,"about_ca_system_score_codex":0.00070091983,"about_ca_system_score_gemma":0.0008207619,"threshold_uncertainty_score":0.011833131},"labels":[],"label_agreement":null},{"id":"W2182573229","doi":"10.1609/aaai.v29i1.9701","title":"Representation Discovery for MDPs Using Bisimulation Metrics","year":2015,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Bisimulation; Metric (unit); Computer science; Representation (politics); Computation; Convergence (economics); Markov decision process; Sequence (biology); Theoretical computer science; State space; State (computer science); Metric space; Algorithm; Markov process; Mathematics; Discrete mathematics","score_opus":0.29689374545324165,"score_gpt":0.37833881418119636,"score_spread":0.0814450687279547,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2182573229","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004125882,0.000042038966,0.99485224,0.00005614967,0.00000708661,0.000057441328,0.00004160669,0.00028719186,0.00053033046],"genre_scores_gemma":[0.16810647,0.00013504483,0.83000755,0.000056497935,0.00001466087,0.00042630982,0.0003721476,0.00019869092,0.0006826429],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99781847,0.0007096104,0.00016754992,0.00040099336,0.0007582337,0.00014513795],"domain_scores_gemma":[0.99545044,0.0027596562,0.00045097,0.00063350296,0.00057413004,0.00013136736],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024974702,0.0013634339,0.0013556224,0.0020483395,0.0008006251,0.0017505758,0.0020111143,0.0012782825,0.0028893366],"category_scores_gemma":[0.0141487075,0.0008458691,0.0018373958,0.0013214478,0.0014201468,0.003278455,0.0037219191,0.0021885275,0.0006628627],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007095638,0.00006233696,0.00069737463,0.00017528643,0.000050755913,0.00008662317,0.00020857429,0.7584973,0.0027920878,0.116284825,0.00094769883,0.12012611],"study_design_scores_gemma":[0.000008481366,0.000022157854,0.000030785413,0.000014579171,0.000005316749,0.000016203474,0.000016287477,0.9647441,0.0010360285,0.033298135,0.0008009404,0.0000070374444],"about_ca_topic_score_codex":0.00426104,"about_ca_topic_score_gemma":0.0042428966,"teacher_disagreement_score":0.00426104,"about_ca_system_score_codex":0.002071491,"about_ca_system_score_gemma":0.00265392,"threshold_uncertainty_score":0.015029728},"labels":[],"label_agreement":null},{"id":"W2182877511","doi":"10.82308/46241","title":"Optimal time scales for reinforcement learning behaviour strategies","year":2010,"lang":"en","type":"article","venue":"Open MIND","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Temporal difference learning; Formalism (music); Gradient descent; Representation (politics); Q-learning; Scale (ratio); Machine learning; Mathematical optimization; Artificial neural network; Mathematics","score_opus":0.026822870449726807,"score_gpt":0.30606543534320313,"score_spread":0.2792425648934763,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2182877511","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.058670636,0.00029951145,0.9313096,0.0004994676,0.000042581232,0.0000902193,0.000051433555,0.00026624266,0.008770403],"genre_scores_gemma":[0.8747287,0.000280984,0.11878933,0.00012011057,0.0000344442,0.00031174105,0.00009231822,0.00014886106,0.0054935385],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991412,0.00031511727,0.00004969918,0.0001655886,0.00021622545,0.00011221181],"domain_scores_gemma":[0.9967873,0.002230201,0.00032659023,0.0001569182,0.00026597144,0.000233109],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018719166,0.00079086696,0.00077679654,0.0005921947,0.0005761854,0.0016203028,0.0009150578,0.0011793621,0.004885925],"category_scores_gemma":[0.0124355145,0.000459238,0.0005642699,0.00028505528,0.0017093836,0.0022378135,0.0015277882,0.0020415443,0.00050867116],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001363442,0.000095516945,0.0008837396,0.00010521083,0.0000313857,0.000086001055,0.00023237197,0.6618337,0.002491856,0.3008042,0.0010005991,0.032299012],"study_design_scores_gemma":[0.000022192322,0.000032028704,0.00010988893,0.000014516459,0.0000057344623,0.000008191371,0.000017346641,0.9272394,0.00038973187,0.07171836,0.00043505064,0.0000076319875],"about_ca_topic_score_codex":0.0018395064,"about_ca_topic_score_gemma":0.0014275808,"teacher_disagreement_score":0.004885925,"about_ca_system_score_codex":0.001906064,"about_ca_system_score_gemma":0.0011961893,"threshold_uncertainty_score":0.016345024},"labels":[],"label_agreement":null},{"id":"W2184461682","doi":"10.82308/33420","title":"A Bayesian Framework for Online Parameter Learning in POMDPs","year":2011,"lang":"en","type":"article","venue":"eScholarship@McGill (McGill)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Partially observable Markov decision process; Computer science; Reinforcement learning; Artificial intelligence; Markov decision process; Machine learning; Ambiguity; Robotics; Bayesian probability; Robot; Markov process; Markov chain; Markov model","score_opus":0.043054267710654175,"score_gpt":0.26167763034968494,"score_spread":0.21862336263903076,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2184461682","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00096681225,0.0003039652,0.99635434,0.00018800818,0.00003092035,0.00003810391,0.00007607148,0.00016812842,0.0018738087],"genre_scores_gemma":[0.31827015,0.0023009833,0.6688927,0.0003457954,0.0003082248,0.0010403153,0.0007054642,0.00029137504,0.007845054],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974655,0.0010045588,0.00016506828,0.00041625949,0.0007288316,0.00021988532],"domain_scores_gemma":[0.9960342,0.0027820983,0.00032984538,0.00017728122,0.00050127675,0.00017527364],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004289894,0.0022521964,0.0024003992,0.0014962814,0.0010947963,0.0032215053,0.003876127,0.0023869402,0.0060987067],"category_scores_gemma":[0.011120272,0.0018493843,0.0022842002,0.0015323529,0.0028850192,0.0040544034,0.0026549792,0.004754013,0.0011214936],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003786779,0.000032726744,0.00025662503,0.00010048842,0.00004044961,0.00007511242,0.00010255925,0.81622934,0.00028055298,0.16481382,0.00092408934,0.017106362],"study_design_scores_gemma":[0.000019427625,0.000021529475,0.00004680283,0.000024784364,0.000010594173,0.000013544997,0.000012451145,0.9239756,0.00010375287,0.07398934,0.0017659565,0.000016314005],"about_ca_topic_score_codex":0.019388568,"about_ca_topic_score_gemma":0.017223194,"teacher_disagreement_score":0.019388568,"about_ca_system_score_codex":0.0034781399,"about_ca_system_score_gemma":0.003624255,"threshold_uncertainty_score":0.03855139},"labels":[],"label_agreement":null},{"id":"W2185087676","doi":"10.13140/2.1.4274.3043","title":"Efficient Planning in MDPs by Small Backups","year":2013,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Backup; Successor cardinal; Computer science; Reinforcement learning; Flexibility (engineering); Computation; Implementation; Process (computing); Distributed computing; Artificial intelligence; Algorithm; Mathematics; Operating system","score_opus":0.016645286899418594,"score_gpt":0.22778139205813944,"score_spread":0.21113610515872083,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2185087676","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0792909,0.00023536554,0.9169083,0.00026940307,0.00004174206,0.00007630168,0.000086342145,0.0011390791,0.0019525294],"genre_scores_gemma":[0.80792093,0.00015676,0.19028331,0.000074971205,0.000020619207,0.00018809638,0.00011838395,0.00012980402,0.0011071377],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994752,0.00017580688,0.000039647774,0.0001042309,0.00012624456,0.00007880075],"domain_scores_gemma":[0.99754864,0.001530171,0.00016439798,0.00047369316,0.0001327251,0.00015038233],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012740272,0.00083137566,0.0010389833,0.00042920947,0.00055446936,0.00072016893,0.0012439745,0.00078554987,0.0029978172],"category_scores_gemma":[0.00506154,0.0005495051,0.0005377639,0.00042859477,0.0011817265,0.0015758197,0.0018463361,0.0016651566,0.00038932113],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002850953,0.0000778362,0.00062661804,0.00009558215,0.000029597122,0.000117739946,0.00012561401,0.9136045,0.004962451,0.015073502,0.0008489408,0.06415249],"study_design_scores_gemma":[0.000039515748,0.000049197464,0.00006512933,0.000007669563,0.0000069687535,0.000021766135,0.000017366605,0.98396975,0.0014295689,0.013917187,0.00047024534,0.000005565531],"about_ca_topic_score_codex":0.0022012168,"about_ca_topic_score_gemma":0.0024090675,"teacher_disagreement_score":0.0029978172,"about_ca_system_score_codex":0.0005472002,"about_ca_system_score_gemma":0.0011159478,"threshold_uncertainty_score":0.01002872},"labels":[],"label_agreement":null},{"id":"W2188892596","doi":"10.13140/2.1.1456.2568","title":"True Online TD(λ)","year":2014,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Equivalence (formal languages); Algorithm; TRACE (psycholinguistics); Function (biology); Simple (philosophy); Reinforcement learning; Online algorithm; Theoretical computer science; Artificial intelligence; Mathematics","score_opus":0.011263924549713058,"score_gpt":0.2308848390242981,"score_spread":0.21962091447458504,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2188892596","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006246683,0.00011139134,0.9865754,0.00024097202,0.000121911085,0.000079642654,0.00012182915,0.0029384845,0.0035636986],"genre_scores_gemma":[0.34217358,0.00013953402,0.6425018,0.00059532153,0.000104127714,0.0003033558,0.00053169514,0.0007365127,0.012914074],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9986743,0.00029642184,0.00008727753,0.00039503505,0.0003680578,0.00017890126],"domain_scores_gemma":[0.9959998,0.0018609383,0.00023499137,0.0010256193,0.0006796284,0.00019901848],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018707055,0.0008401694,0.0008015399,0.000428549,0.0005042746,0.0016347979,0.0027205856,0.0015356453,0.010956807],"category_scores_gemma":[0.009056724,0.00041274677,0.0006028966,0.00046014597,0.000987036,0.0022906424,0.0021298584,0.0023699878,0.0026360394],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005248004,0.00029013533,0.001679446,0.0002846593,0.000060963863,0.000129813,0.0001782418,0.1645452,0.004530951,0.089847706,0.021025684,0.7169024],"study_design_scores_gemma":[0.00008126262,0.000102772996,0.00018839067,0.00002395356,0.000017268856,0.00012028142,0.000031157248,0.9356537,0.005286029,0.049541052,0.0089323865,0.000021800275],"about_ca_topic_score_codex":0.004604914,"about_ca_topic_score_gemma":0.0054058363,"teacher_disagreement_score":0.010956807,"about_ca_system_score_codex":0.0013139097,"about_ca_system_score_gemma":0.0034988553,"threshold_uncertainty_score":0.036654115},"labels":[],"label_agreement":null},{"id":"W2190606234","doi":"10.5555/2936924.2936996","title":"State of the Art Control of Atari Games Using Shallow Reinforcement Learning","year":2016,"lang":"en","type":"article","venue":"Adaptive Agents and Multi-Agents Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":56,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Benchmark (surveying); Representation (politics); Artificial intelligence; Set (abstract data type); Strengths and weaknesses; Simple (philosophy); Key (lock); Artificial neural network; Feature learning; Machine learning","score_opus":0.04777979186199204,"score_gpt":0.2658761195988801,"score_spread":0.21809632773688803,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2190606234","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0728725,0.00096091017,0.90982074,0.00041075543,0.00010755236,0.00011392329,0.00007527644,0.00073953747,0.014898789],"genre_scores_gemma":[0.9447827,0.00024121953,0.052100804,0.00010378098,0.000030788564,0.00009298831,0.00007215355,0.00004738665,0.0025282798],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99959856,0.00010514635,0.000026071157,0.00008634796,0.00011326415,0.000070497066],"domain_scores_gemma":[0.9990103,0.00056357746,0.00009765029,0.00010948545,0.00014263153,0.00007621785],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010111791,0.0009839527,0.00085793866,0.00029504037,0.00034110053,0.0009490675,0.0015941338,0.0008724889,0.0030845879],"category_scores_gemma":[0.002531181,0.00034086744,0.00048682466,0.0001886713,0.0010537781,0.00092467625,0.0013389668,0.0014388298,0.00035725883],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000112405716,0.00007172674,0.0004920569,0.00010272667,0.00004070289,0.000036393292,0.000046541205,0.9276539,0.0014079239,0.013765042,0.00074277364,0.055527862],"study_design_scores_gemma":[0.000008784979,0.000027235303,0.000037569447,0.0000045203005,0.0000027426552,0.000003028125,0.000002231188,0.9974535,0.0002005523,0.0020579374,0.00019943806,0.000002355564],"about_ca_topic_score_codex":0.007398173,"about_ca_topic_score_gemma":0.0071498505,"teacher_disagreement_score":0.007398173,"about_ca_system_score_codex":0.0008763081,"about_ca_system_score_gemma":0.0009967914,"threshold_uncertainty_score":0.0147102475},"labels":[],"label_agreement":null},{"id":"W2191938486","doi":"","title":"Basis refinement strategies for linear value function approximation in MDPs","year":2015,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Markov decision process; Basis (linear algebra); Computer science; Bellman equation; Bisimulation; Basis function; Mathematical optimization; Function (biology); Value (mathematics); Markov process; Algorithm; Theoretical computer science; Mathematics; Machine learning","score_opus":0.05119097121540838,"score_gpt":0.2817875639405395,"score_spread":0.2305965927251311,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2191938486","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005500892,0.0001117093,0.9931767,0.00007897606,0.000008082738,0.000029882542,0.000013900503,0.00006507704,0.0010147325],"genre_scores_gemma":[0.48055756,0.0005287553,0.5150491,0.00014149975,0.00003530925,0.00058386667,0.00016395688,0.00014930607,0.0027906077],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99852294,0.0007016817,0.00007240925,0.00016288602,0.00040359877,0.0001365501],"domain_scores_gemma":[0.99559826,0.0032636793,0.00022475187,0.00036067062,0.00040461283,0.00014809208],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003997158,0.0008676283,0.0014144941,0.0009974341,0.000618304,0.0011800274,0.0017324853,0.0013342128,0.0027378506],"category_scores_gemma":[0.013598277,0.000559744,0.0010220882,0.0009199546,0.0019933071,0.0022307243,0.0028011228,0.0025896821,0.0004344501],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005438016,0.000045842986,0.0003256291,0.000089906556,0.000035149635,0.000040732935,0.00012763329,0.6509156,0.0011964298,0.31130597,0.0005792712,0.035283495],"study_design_scores_gemma":[0.000008303163,0.000019995714,0.000020965881,0.0000119246115,0.000004121448,0.0000058645446,0.000008808714,0.94096196,0.00030356488,0.0583292,0.00032080323,0.0000045344664],"about_ca_topic_score_codex":0.0025779523,"about_ca_topic_score_gemma":0.002156374,"teacher_disagreement_score":0.003997158,"about_ca_system_score_codex":0.0018725947,"about_ca_system_score_gemma":0.0014359328,"threshold_uncertainty_score":0.021139264},"labels":[],"label_agreement":null},{"id":"W2197326744","doi":"10.1017/s0269888910000354","title":"Robotics competitions as benchmarks for AI research","year":2011,"lang":"en","type":"article","venue":"The Knowledge Engineering Review","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Robotics; Benchmark (surveying); Artificial intelligence; Computer science; Applications of artificial intelligence; Robot","score_opus":0.1069341295820813,"score_gpt":0.3609366505106727,"score_spread":0.2540025209285914,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2197326744","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.084843196,0.3374326,0.014101933,0.09013498,0.014187058,0.0004680233,0.0012036001,0.00029372398,0.4573349],"genre_scores_gemma":[0.8602203,0.0717147,0.01385811,0.015250677,0.0057574357,0.00067800493,0.0023986753,0.00030235396,0.029819703],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98068225,0.010601411,0.0007933834,0.0008917141,0.0061223023,0.00090903795],"domain_scores_gemma":[0.9614685,0.016025322,0.003519653,0.0013400061,0.01464217,0.0030043337],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.028983703,0.0005800533,0.0009266664,0.005247228,0.0016759663,0.005217546,0.002015297,0.0015755923,0.007618967],"category_scores_gemma":[0.033703912,0.00019724124,0.0003713329,0.0049308487,0.0026035332,0.0038743971,0.003065137,0.0018762008,0.0016393219],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007215535,0.00047170257,0.0043841633,0.0020926567,0.0001496134,0.00014128807,0.0008038656,0.005258766,0.0007261587,0.45664608,0.17471267,0.35389155],"study_design_scores_gemma":[0.0002494017,0.0008991752,0.015707381,0.0025990568,0.0000748347,0.00019972294,0.0029263466,0.0049592406,0.0017459239,0.13679343,0.83371824,0.00012723515],"about_ca_topic_score_codex":0.004005329,"about_ca_topic_score_gemma":0.0062035257,"teacher_disagreement_score":0.9710163,"about_ca_system_score_codex":0.005317928,"about_ca_system_score_gemma":0.003595746,"threshold_uncertainty_score":0.15328228},"labels":[],"label_agreement":null},{"id":"W2198357760","doi":"10.1609/aaai.v29i1.9702","title":"Reward Shaping for Model-Based Bayesian Reinforcement Learning","year":2015,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Defense Acquisition Program Administration; Agency for Defense Development; National Research Foundation of Korea; Ministry of Science, ICT and Future Planning; National Research Foundation","keywords":"Reinforcement learning; Benchmark (surveying); Machine learning; Computer science; Artificial intelligence; Heuristic; Bayesian probability; Function (biology); Bayes' theorem; Domain (mathematical analysis); Bayesian inference; Mathematics","score_opus":0.16860257365642317,"score_gpt":0.32334944967993895,"score_spread":0.15474687602351578,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2198357760","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008320238,0.00027505297,0.98756266,0.00028474434,0.000029429722,0.000045172612,0.0000427924,0.00032734545,0.0031126388],"genre_scores_gemma":[0.8441398,0.00036882216,0.15186793,0.00030071742,0.00005033469,0.00031663102,0.00012438605,0.00014809941,0.0026833278],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99833965,0.0008420249,0.00006443527,0.00022058317,0.0003854898,0.00014784191],"domain_scores_gemma":[0.99579704,0.0029666494,0.00035437912,0.00029404092,0.0003758431,0.00021217],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034490433,0.0011810698,0.0014678822,0.0006258523,0.00045062474,0.0012694382,0.0017505992,0.00150083,0.0033399896],"category_scores_gemma":[0.015234772,0.0005842118,0.0005463179,0.00053673505,0.0017955201,0.0019831897,0.0019520695,0.0023843946,0.0005436678],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000072616946,0.000054761167,0.00028477848,0.00007880075,0.000022594304,0.000035707522,0.00006095665,0.90781164,0.00069530477,0.06605547,0.0009487285,0.023878584],"study_design_scores_gemma":[0.000011556389,0.000018280181,0.000022289461,0.000008070245,0.000002841245,0.0000056715567,0.0000028499237,0.9739354,0.00012367062,0.025595948,0.00026888365,0.0000044654407],"about_ca_topic_score_codex":0.0026384061,"about_ca_topic_score_gemma":0.00211786,"teacher_disagreement_score":0.0034490433,"about_ca_system_score_codex":0.001987776,"about_ca_system_score_gemma":0.0018477928,"threshold_uncertainty_score":0.018240452},"labels":[],"label_agreement":null},{"id":"W2201672140","doi":"10.1609/aaai.v25i1.7918","title":"Basis Function Discovery Using Spectral Clustering and Bisimulation Metrics","year":2011,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Office of Naval Research; Fonds Québécois de la Recherche sur la Nature et les Technologies","keywords":"Adjacency list; Cluster analysis; Computer science; Basis (linear algebra); Graph; Function (biology); Theoretical computer science; Spectral clustering; State (computer science); Artificial intelligence; Data mining; Algorithm; Mathematics","score_opus":0.1564952070208593,"score_gpt":0.2914427844955239,"score_spread":0.1349475774746646,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2201672140","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028121827,0.00015764737,0.97047645,0.00016192824,0.000012597563,0.00004538521,0.000033310305,0.00025139202,0.0007394878],"genre_scores_gemma":[0.5536378,0.0002236377,0.44408962,0.00010718973,0.000037408838,0.00025152057,0.00025416393,0.00022674089,0.0011719711],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99738103,0.0012554694,0.00013373057,0.00044363394,0.0006172904,0.0001688077],"domain_scores_gemma":[0.98465043,0.010901903,0.0014246416,0.0012969628,0.0012567324,0.00046935136],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004374448,0.0012711593,0.002015749,0.0028758228,0.0009847324,0.0016788793,0.0020011484,0.002089462,0.0018109833],"category_scores_gemma":[0.027384205,0.00081845326,0.0012116255,0.0017149746,0.0020895316,0.0034753329,0.0029415372,0.0020345692,0.0004702256],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014252671,0.00013850823,0.0018074693,0.00014083707,0.000072060386,0.0000700055,0.00020434141,0.82847255,0.0021186431,0.07948257,0.00088125304,0.086469226],"study_design_scores_gemma":[0.000006685411,0.000019933688,0.000055330078,0.0000070320557,0.000003520858,0.000009685911,0.0000073351457,0.97740424,0.00043317952,0.021922978,0.00012434094,0.0000057973784],"about_ca_topic_score_codex":0.0025537799,"about_ca_topic_score_gemma":0.0017774547,"teacher_disagreement_score":0.004374448,"about_ca_system_score_codex":0.002010852,"about_ca_system_score_gemma":0.0014437484,"threshold_uncertainty_score":0.02313453},"labels":[],"label_agreement":null},{"id":"W2205300805","doi":"10.1609/aiide.v9i1.12609","title":"Scythe AI: A Tool for Modular Reuse of Game AI","year":2013,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Modular design; Reuse; Computer science; Software engineering; Process (computing); Artificial intelligence; Programming language; Human–computer interaction; Engineering","score_opus":0.0330847396573221,"score_gpt":0.2804623404342777,"score_spread":0.2473776007769556,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2205300805","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0036046004,0.000086917855,0.9225539,0.00012331789,0.000063923966,0.00020961642,0.00036703178,0.0633868,0.009603934],"genre_scores_gemma":[0.102296464,0.00038742487,0.85091764,0.00032119476,0.000050161776,0.0009431305,0.003379894,0.020195551,0.021508656],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987853,0.00023224708,0.00010315204,0.00018631098,0.00053655845,0.0001565088],"domain_scores_gemma":[0.9971318,0.0014084821,0.000120923636,0.0007922152,0.00032645397,0.00022013065],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002062563,0.0013578682,0.00058737723,0.0018628086,0.0007378043,0.0024992016,0.0031051254,0.00131817,0.021143744],"category_scores_gemma":[0.0074531897,0.0012000501,0.0014762558,0.0006785978,0.0017564989,0.0040609455,0.00525643,0.002780769,0.0056924107],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078175566,0.00057090004,0.003149678,0.0016254892,0.000223378,0.0018221985,0.0046920306,0.047336034,0.05122857,0.22019951,0.118696146,0.5496742],"study_design_scores_gemma":[0.00038105185,0.0002760163,0.0012199957,0.0003833891,0.00010819174,0.0018280043,0.00036413138,0.285206,0.056120235,0.091715835,0.5621454,0.00025175037],"about_ca_topic_score_codex":0.0023164183,"about_ca_topic_score_gemma":0.0028492971,"teacher_disagreement_score":0.021143744,"about_ca_system_score_codex":0.0007577029,"about_ca_system_score_gemma":0.001302142,"threshold_uncertainty_score":0.07073283},"labels":[],"label_agreement":null},{"id":"W2219400200","doi":"10.1609/aaai.v29i1.9655","title":"Approximate Linear Programming for Constrained Partially Observable Markov Decision Processes","year":2015,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":60,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; University of Waterloo","funders":"Institute for Information and Communications Technology Promotion; Natural Sciences and Engineering Research Council of Canada; Ministry of Science, ICT and Future Planning","keywords":"Observable; Partially observable Markov decision process; Markov decision process; Mathematical optimization; Benchmark (surveying); Computer science; Linear programming; Suite; Sequence (biology); Dynamic programming; Controller (irrigation); State (computer science); Markov process; Mathematics; Algorithm","score_opus":0.14263682547995002,"score_gpt":0.3262468132388472,"score_spread":0.18360998775889717,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2219400200","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0135177085,0.0005408189,0.9822875,0.00032589203,0.000025788848,0.00005301585,0.00009621424,0.00028869158,0.0028642707],"genre_scores_gemma":[0.76715064,0.0007324531,0.22465959,0.00022755076,0.000060372342,0.000511608,0.00035818535,0.000157186,0.0061424645],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987923,0.00059713254,0.000042853007,0.00018746831,0.0002543723,0.00012588692],"domain_scores_gemma":[0.9952413,0.004040951,0.00028727675,0.00010686628,0.0002346454,0.00008902016],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018151583,0.0011392457,0.0014493245,0.00048763838,0.00040425945,0.0013001491,0.0008525918,0.0012335435,0.0036294677],"category_scores_gemma":[0.007118206,0.0007363043,0.0006227719,0.000820516,0.0013383579,0.0011153595,0.0011318108,0.002003606,0.00034293806],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002965371,0.000016368791,0.00011230312,0.000038614107,0.000013626123,0.000024101693,0.000019873634,0.9809651,0.00014296795,0.013259006,0.00024638698,0.0051320004],"study_design_scores_gemma":[0.0000044890726,0.0000050093936,0.000013040314,0.0000026356595,0.0000013166043,0.0000018472043,0.000002398013,0.99177176,0.00004503551,0.008053102,0.00009818403,0.0000011479783],"about_ca_topic_score_codex":0.009674829,"about_ca_topic_score_gemma":0.009023144,"teacher_disagreement_score":0.009674829,"about_ca_system_score_codex":0.0018580297,"about_ca_system_score_gemma":0.002162361,"threshold_uncertainty_score":0.019236982},"labels":[],"label_agreement":null},{"id":"W2232305727","doi":"10.1609/aaai.v29i1.9700","title":"Tighter Value Function Bounds for Bayesian Reinforcement Learning","year":2015,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Defense Acquisition Program Administration; Agency for Defense Development; National Research Foundation of Korea; Ministry of Science, ICT and Future Planning; National Research Foundation","keywords":"Reinforcement learning; Bayesian probability; Computer science; Heuristic; Bellman equation; Bayes' theorem; Function (biology); Perspective (graphical); Value (mathematics); Artificial intelligence; Machine learning; Mathematical optimization; Focus (optics); Mathematics","score_opus":0.09005187973659068,"score_gpt":0.2974173093581666,"score_spread":0.20736542962157595,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2232305727","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004064534,0.0004512745,0.9918521,0.00039569577,0.00004184445,0.00004397385,0.00003831208,0.00020475997,0.0029074575],"genre_scores_gemma":[0.5231126,0.0013004367,0.46856558,0.0009903042,0.00028312113,0.00084253866,0.00032627909,0.0007196479,0.0038595926],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9921227,0.0038148204,0.00035912718,0.0009361127,0.002098082,0.00066921185],"domain_scores_gemma":[0.9574349,0.035783462,0.0017695787,0.0020829088,0.0021423341,0.00078689767],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013951349,0.0025839144,0.0032706312,0.002541065,0.0012158962,0.004452459,0.0029866048,0.0034800062,0.0062312884],"category_scores_gemma":[0.087317325,0.001537704,0.0015551413,0.0017740306,0.004484632,0.00834672,0.005587924,0.009845499,0.0011793532],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009496891,0.00010353884,0.00045328742,0.00015295614,0.000052003645,0.00005051314,0.00016852493,0.7046476,0.0008358313,0.26319414,0.0016427421,0.028603863],"study_design_scores_gemma":[0.00001548075,0.000021561875,0.000048227706,0.000043470947,0.000007056229,0.00000914058,0.000010041664,0.86354184,0.00030260655,0.13547018,0.00051963946,0.000010809882],"about_ca_topic_score_codex":0.0035197006,"about_ca_topic_score_gemma":0.0033167703,"teacher_disagreement_score":0.013951349,"about_ca_system_score_codex":0.0053103734,"about_ca_system_score_gemma":0.0035410176,"threshold_uncertainty_score":0.07378268},"labels":[],"label_agreement":null},{"id":"W223326216","doi":"10.1613/jair.4301","title":"Policy Iteration Based on Stochastic Factorization","year":2014,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; National Institutes of Health; McGill University; Fonds Québécois de la Recherche sur la Nature et les Technologies; Compute Canada; Coordenação de Aperfeiçoamento de Pessoal de Nível Superior","keywords":"Markov decision process; Factorization; Mathematical optimization; Factoring; Mathematics; Multiplication (music); Computer science; Markov process; Applied mathematics; Algorithm; Finance","score_opus":0.12763786672913804,"score_gpt":0.4165855694153244,"score_spread":0.2889477026861863,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W223326216","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0058714757,0.00006615995,0.99257267,0.00007912912,0.000021979293,0.000039237108,0.000016844187,0.00017396764,0.0011584837],"genre_scores_gemma":[0.55748194,0.00024281237,0.4389271,0.00013374416,0.000055106,0.00033462743,0.00014222295,0.00009561291,0.002586831],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988035,0.0004441836,0.0000665467,0.00022260283,0.00031235637,0.00015079006],"domain_scores_gemma":[0.9972832,0.0019513549,0.00020391872,0.0001756597,0.0002864528,0.00009949982],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017025069,0.0009004488,0.001511597,0.00059927366,0.0005271062,0.00092552765,0.0008708689,0.0010659427,0.0025940163],"category_scores_gemma":[0.0059756604,0.00052158954,0.000722343,0.00053267466,0.0013798566,0.0012849359,0.0012639824,0.0015067712,0.00039737084],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009184562,0.00003669708,0.00044533523,0.0000694734,0.000029773637,0.000073348725,0.0000816694,0.90189177,0.0014725075,0.04735608,0.00071319955,0.04773841],"study_design_scores_gemma":[0.000008803598,0.00001901004,0.000019331626,0.000005059947,0.000002517949,0.0000136368135,0.0000034707625,0.9889543,0.0003645056,0.010305468,0.00030053136,0.0000033939757],"about_ca_topic_score_codex":0.0054687923,"about_ca_topic_score_gemma":0.003335363,"teacher_disagreement_score":0.0054687923,"about_ca_system_score_codex":0.0012446536,"about_ca_system_score_gemma":0.002549771,"threshold_uncertainty_score":0.010873914},"labels":[],"label_agreement":null},{"id":"W2236244207","doi":"10.1561/2200000049","title":"Bayesian Reinforcement Learning: A Survey","year":2015,"lang":"en","type":"article","venue":"Foundations and Trends® in Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":223,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Machine learning; Reinforcement learning; Artificial intelligence; Bayesian inference; Bayesian probability; Variable-order Bayesian network; Prior probability; Inference; Bellman equation; Algorithm; Mathematical optimization; Mathematics","score_opus":0.0465390705497981,"score_gpt":0.3022379049415592,"score_spread":0.2556988343917611,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2236244207","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001584646,0.22614485,0.7485257,0.0038411391,0.0005675985,0.000108012195,0.00026075018,0.00040691695,0.018560458],"genre_scores_gemma":[0.09047037,0.5075034,0.3839407,0.0018977427,0.003980792,0.00064238516,0.000834936,0.00049035583,0.010239395],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9968479,0.0013230108,0.00021224265,0.0004120325,0.0010979474,0.000106842876],"domain_scores_gemma":[0.99040353,0.007986934,0.00022033197,0.00042866453,0.0008227439,0.00013776273],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004878199,0.0018061531,0.002751805,0.0023390583,0.00068841054,0.0035150975,0.0029223256,0.002894475,0.006918041],"category_scores_gemma":[0.015099551,0.0013078413,0.0012792842,0.004347405,0.0020176144,0.004233444,0.0019997377,0.0037369058,0.0029499833],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006478524,0.00013900148,0.0012817653,0.0022745081,0.00015921483,0.00006810849,0.0001853246,0.053501356,0.00031253567,0.34460476,0.018717436,0.57869124],"study_design_scores_gemma":[0.00006125388,0.00009670688,0.0009556268,0.0014011755,0.000091394395,0.00030029318,0.00009830664,0.21353579,0.00055442465,0.5736824,0.20912783,0.00009489838],"about_ca_topic_score_codex":0.005603018,"about_ca_topic_score_gemma":0.0040093698,"teacher_disagreement_score":0.006918041,"about_ca_system_score_codex":0.0023827392,"about_ca_system_score_gemma":0.0032104251,"threshold_uncertainty_score":0.025798678},"labels":[],"label_agreement":null},{"id":"W2238086212","doi":"10.15307/fcj.25.185.2015","title":"FCJ-185 An Algorithmic Agartha: Post-App Approaches to Synarchic Regulation","year":2015,"lang":"en","type":"article","venue":"The Fibreculture Journal","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science; World Wide Web; Internet privacy","score_opus":0.13523283284432092,"score_gpt":0.2589250793733536,"score_spread":0.12369224652903266,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2238086212","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023175796,0.002110589,0.16207057,0.026954256,0.00077469164,0.00007310439,0.0001197483,0.00025557657,0.7844656],"genre_scores_gemma":[0.8642413,0.0011654223,0.03693211,0.003462602,0.00052313454,0.0002890146,0.00008864285,0.00023395542,0.09306376],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9979159,0.0008585716,0.000083532825,0.00046687655,0.0004325061,0.00024252833],"domain_scores_gemma":[0.99770653,0.0010004976,0.00019013876,0.00059963454,0.00035783523,0.00014527538],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027321205,0.0004766006,0.0003437734,0.0009894582,0.0032034363,0.006577531,0.0012674412,0.0024753474,0.010998648],"category_scores_gemma":[0.007414593,0.00027635728,0.00055119814,0.00084991317,0.019290406,0.007545362,0.002716464,0.003944043,0.0014534752],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000024844214,0.0000030461315,0.000035951824,0.000006398403,0.0000011394269,0.000008043723,0.00032537605,0.00028059975,0.000029809204,0.99616325,0.00084116176,0.0023025875],"study_design_scores_gemma":[0.00000430896,0.000007934661,0.00015064387,0.000022953269,0.0000020641223,0.000023652203,0.00020961677,0.0016609298,0.00010601375,0.9547579,0.04304634,0.000007538956],"about_ca_topic_score_codex":0.0057294234,"about_ca_topic_score_gemma":0.0036978582,"teacher_disagreement_score":0.010998648,"about_ca_system_score_codex":0.0055930587,"about_ca_system_score_gemma":0.002472843,"threshold_uncertainty_score":0.04058069},"labels":[],"label_agreement":null},{"id":"W2264982354","doi":"10.3166/ria.27.243-263","title":"Adaptation de la matrice de covariance pour l’apprentissage par renforcement direct","year":2013,"lang":"fr","type":"article","venue":"Revue d intelligence artificielle","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Adaptation (eye); Reinforcement learning; Covariance; Parameterized complexity; Humanities; Political science; Computer science; Psychology; Artificial intelligence; Mathematics; Philosophy; Algorithm; Neuroscience; Statistics","score_opus":0.04328650558652317,"score_gpt":0.278600095215143,"score_spread":0.23531358962861984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2264982354","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018208949,0.0002959595,0.9780806,0.0004215392,0.00019002959,0.0000866589,0.000043314558,0.000562049,0.0021110123],"genre_scores_gemma":[0.62215656,0.0006779091,0.3428693,0.0005761361,0.00043825863,0.00073502946,0.00025091873,0.00050381676,0.031792082],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99788624,0.00088830077,0.00008995988,0.00048061425,0.00047927565,0.00017558096],"domain_scores_gemma":[0.99271023,0.0052189063,0.00020659805,0.00055059086,0.0010949213,0.00021874547],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034220128,0.0013684572,0.0016174851,0.0004956106,0.00056889706,0.0014825729,0.001302782,0.0030558535,0.0066118557],"category_scores_gemma":[0.0124203805,0.00066734274,0.0012921982,0.00042051647,0.0015586459,0.0014599687,0.0013737006,0.0036724603,0.0017500628],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004234938,0.00020313177,0.000714266,0.00014133588,0.00013442153,0.00015529312,0.00018189887,0.8201524,0.015493109,0.021379104,0.0031692823,0.13785236],"study_design_scores_gemma":[0.000023766039,0.00005037642,0.00033903087,0.000008070872,0.0000065488316,0.000022647455,0.0000059047647,0.9958812,0.0017636914,0.0012035868,0.0006815315,0.000013636824],"about_ca_topic_score_codex":0.026006186,"about_ca_topic_score_gemma":0.01932293,"teacher_disagreement_score":0.026006186,"about_ca_system_score_codex":0.001460388,"about_ca_system_score_gemma":0.0026524097,"threshold_uncertainty_score":0.051709652},"labels":[],"label_agreement":null},{"id":"W2269274350","doi":"10.48550/arxiv.1512.04087","title":"True Online Temporal-Difference Learning","year":2015,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":56,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Temporal difference learning; Computer science; Artificial intelligence; Equivalence (formal languages); Lambda; Reinforcement learning; Domain (mathematical analysis); Online learning; Machine learning; Binary number; Online algorithm; Simple (philosophy); Algorithm; Theoretical computer science; Mathematics; Discrete mathematics; Arithmetic; Mathematical analysis","score_opus":0.12191799175305862,"score_gpt":0.2099222055253362,"score_spread":0.08800421377227757,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2269274350","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037114598,0.0007083499,0.9532283,0.0005286254,0.0002887548,0.000111006426,0.000260054,0.0028638395,0.0048964345],"genre_scores_gemma":[0.64985204,0.00024388266,0.34155717,0.00067510747,0.00013500206,0.0002608075,0.0006469206,0.00036919708,0.0062599266],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99809617,0.00062620075,0.00012641323,0.0004974932,0.0004713626,0.0001824226],"domain_scores_gemma":[0.9923006,0.0044418476,0.0004126986,0.0014827347,0.0010170018,0.00034509483],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038872582,0.0011677075,0.0014480912,0.00055045076,0.00042385,0.0013355884,0.003230354,0.0020113527,0.006732861],"category_scores_gemma":[0.0137805175,0.00046887124,0.0009079475,0.0004866713,0.0013074976,0.0033447186,0.0023515117,0.0029519293,0.0015328951],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010300127,0.00053551374,0.0029348882,0.00041717486,0.00015439931,0.00012394272,0.00013542973,0.50880116,0.00391915,0.033985447,0.0096267145,0.43833613],"study_design_scores_gemma":[0.000041742038,0.00009924239,0.00014239289,0.0000107520855,0.000009351058,0.000033755154,0.000010518199,0.98913026,0.0012227114,0.008626433,0.000663515,0.0000092972605],"about_ca_topic_score_codex":0.0028348714,"about_ca_topic_score_gemma":0.0031750374,"teacher_disagreement_score":0.006732861,"about_ca_system_score_codex":0.0011310208,"about_ca_system_score_gemma":0.0021531414,"threshold_uncertainty_score":0.022523642},"labels":[],"label_agreement":null},{"id":"W2272644799","doi":"10.1609/aaai.v29i1.9659","title":"Information Gathering and Reward Exploitation of Subgoals for POMDPs","year":2015,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; Fonds Québécois de la Recherche sur la Nature et les Technologies","keywords":"Partially observable Markov decision process; Computer science; Benchmark (surveying); Solver; Time horizon; Markov decision process; Space (punctuation); State (computer science); Artificial intelligence; Machine learning; Markov chain; Markov process; Markov model; Mathematical optimization; Mathematics; Algorithm","score_opus":0.1196532880799096,"score_gpt":0.3033813658153713,"score_spread":0.18372807773546168,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2272644799","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036477037,0.00025648525,0.95982593,0.00026921905,0.000025125091,0.00016411502,0.00008704312,0.0008003504,0.0020946055],"genre_scores_gemma":[0.6215093,0.00022186656,0.37633288,0.00014363934,0.000024859477,0.00036910063,0.00021261125,0.00014641203,0.001039306],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99905926,0.00038562383,0.00005812917,0.0001876082,0.00019949907,0.000109892724],"domain_scores_gemma":[0.9974058,0.0018923879,0.0002707098,0.0001670195,0.00013522801,0.00012884979],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019958776,0.001116921,0.0011089774,0.00060199946,0.00056896324,0.0007018175,0.0013090346,0.0009490335,0.0018501705],"category_scores_gemma":[0.005780852,0.00056215265,0.0009824305,0.00041032326,0.0013567448,0.0013926333,0.0018860616,0.0017896652,0.00025665213],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012804203,0.00008767997,0.0009535943,0.00017088484,0.00004493605,0.00011002064,0.00016369001,0.92977196,0.002308069,0.020790692,0.00075447274,0.04471598],"study_design_scores_gemma":[0.000029698193,0.000040285096,0.00007960468,0.000011791424,0.000009376249,0.000013757085,0.000015480773,0.9870129,0.00088445557,0.011427252,0.00046960852,0.0000059096697],"about_ca_topic_score_codex":0.0027482703,"about_ca_topic_score_gemma":0.0032345979,"teacher_disagreement_score":0.0027482703,"about_ca_system_score_codex":0.0009202149,"about_ca_system_score_gemma":0.0018465961,"threshold_uncertainty_score":0.010555327},"labels":[],"label_agreement":null},{"id":"W2276615213","doi":"","title":"Two perspectives on learning rich representations from robot experience","year":2013,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Robot; Artificial intelligence; Computer science; Space (punctuation); Robot learning; Position (finance); Social robot; Human–computer interaction; Mobile robot; Robot control","score_opus":0.021885736796746522,"score_gpt":0.29417326471106175,"score_spread":0.2722875279143152,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2276615213","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0061968244,0.0014275025,0.97149175,0.0063990653,0.00012700375,0.00003781443,0.0002273785,0.00027153056,0.01382113],"genre_scores_gemma":[0.56326205,0.004312802,0.41934764,0.0018229294,0.00077076454,0.0004447802,0.00089518784,0.00024534602,0.00889854],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980567,0.00092076854,0.00008733426,0.0004082801,0.00039075097,0.00013618573],"domain_scores_gemma":[0.99070334,0.0059097027,0.0005764143,0.0018822605,0.0004694875,0.0004587851],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029184031,0.001310484,0.0009857251,0.0015752721,0.0006176117,0.0045928215,0.0032965387,0.002688911,0.0064594033],"category_scores_gemma":[0.012689556,0.001047921,0.0014437105,0.0010602413,0.006514508,0.014072966,0.005121464,0.004990536,0.0009356831],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011716127,0.00008182935,0.000538158,0.0002532598,0.000065075874,0.00009294454,0.000724071,0.03180386,0.0008028808,0.924079,0.0015725415,0.039869145],"study_design_scores_gemma":[0.00002658032,0.000065025626,0.00020624347,0.00006578378,0.000012101952,0.00006246515,0.0000928262,0.04911074,0.0006458368,0.9450561,0.0046260385,0.000030356287],"about_ca_topic_score_codex":0.0010718855,"about_ca_topic_score_gemma":0.00096446264,"teacher_disagreement_score":0.0064594033,"about_ca_system_score_codex":0.0011509039,"about_ca_system_score_gemma":0.00079174497,"threshold_uncertainty_score":0.02160889},"labels":[],"label_agreement":null},{"id":"W2284456400","doi":"10.1007/978-3-642-29946-9_21","title":"A Framework for Computing Bounds for the Return of a Policy","year":2012,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Lipschitz continuity; Computer science; Piecewise; Bounding overwatch; Bounded function; Constant (computer programming); Context (archaeology); Markov decision process; Representation (politics); State (computer science); Theoretical computer science; Mathematical optimization; Markov process; Algorithm; Mathematics; Artificial intelligence; Pure mathematics","score_opus":0.03357722234187318,"score_gpt":0.3053615565312778,"score_spread":0.2717843341894046,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2284456400","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001336708,0.000280578,0.9969831,0.000090008754,0.000042557367,0.000019850746,0.000045598434,0.00039460632,0.0008070096],"genre_scores_gemma":[0.15526628,0.0008833462,0.83850735,0.00020754874,0.00032875597,0.00037634408,0.0004133889,0.00053419697,0.0034827073],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9942644,0.0017137635,0.0004105001,0.0012444332,0.0016265566,0.00074037863],"domain_scores_gemma":[0.982287,0.013043123,0.00094985263,0.0018319263,0.0011558719,0.0007321799],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0101970015,0.002992193,0.0036860015,0.0038491895,0.0015559619,0.00687465,0.006159892,0.004610756,0.007327657],"category_scores_gemma":[0.03514795,0.0024583614,0.003140354,0.0034308902,0.004181515,0.008012778,0.006278137,0.007742045,0.0015105343],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026886456,0.000098134784,0.00046372702,0.00018094075,0.00012566458,0.00007884562,0.00015851637,0.59661114,0.0014014331,0.3121282,0.0034088667,0.08507569],"study_design_scores_gemma":[0.00002390508,0.000047707523,0.00006168374,0.0000354151,0.00002670841,0.000020233952,0.000011670266,0.83373946,0.0005016048,0.16419354,0.0013121095,0.000025885614],"about_ca_topic_score_codex":0.007943181,"about_ca_topic_score_gemma":0.0065069017,"teacher_disagreement_score":0.0101970015,"about_ca_system_score_codex":0.0045679067,"about_ca_system_score_gemma":0.0036208392,"threshold_uncertainty_score":0.05392754},"labels":[],"label_agreement":null},{"id":"W2286609364","doi":"10.5555/2034396.2034427","title":"Basis function discovery using spectral clustering and bisimulation metrics","year":2011,"lang":"en","type":"article","venue":"Adaptive Agents and Multi-Agents Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Markov decision process; Cluster analysis; Set (abstract data type); Bellman equation; State space; Feature (linguistics); Function (biology); Artificial intelligence; State (computer science); Markov process; Feature vector; Basis (linear algebra); Markov chain; Focus (optics); Machine learning; Quality (philosophy); Mathematical optimization; Algorithm; Mathematics","score_opus":0.14918220564433027,"score_gpt":0.28652763876973,"score_spread":0.1373454331253997,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2286609364","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017226772,0.00042515542,0.97969466,0.0002559248,0.000024246408,0.00009220596,0.00008836987,0.00030352373,0.0018891067],"genre_scores_gemma":[0.46357176,0.00075595133,0.5314435,0.00015537888,0.000065869026,0.0005133514,0.00093957986,0.00032754987,0.002227048],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99801683,0.00092191354,0.0001177302,0.00030253222,0.00051317574,0.00012785941],"domain_scores_gemma":[0.9938029,0.0035754812,0.0006128433,0.00062136434,0.0011144534,0.00027291692],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033366613,0.0013584102,0.0020306103,0.0046078092,0.0012519897,0.0020466996,0.0019488661,0.0019986834,0.00260717],"category_scores_gemma":[0.018847529,0.000780482,0.0014547172,0.0026875495,0.0013375627,0.0029587438,0.0029086454,0.0017422837,0.0008688978],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013645238,0.00019694598,0.0021177363,0.00021749284,0.00013735573,0.000080252175,0.00019300393,0.68938357,0.0014942057,0.1422291,0.0039264713,0.15988746],"study_design_scores_gemma":[0.0000068944323,0.00001253597,0.000086451444,0.000012840072,0.0000046019504,0.000012737525,0.0000145039485,0.9622932,0.00025820613,0.036820225,0.00046998757,0.000007756958],"about_ca_topic_score_codex":0.005548796,"about_ca_topic_score_gemma":0.0036723327,"teacher_disagreement_score":0.005548796,"about_ca_system_score_codex":0.002113881,"about_ca_system_score_gemma":0.002238985,"threshold_uncertainty_score":0.017646194},"labels":[],"label_agreement":null},{"id":"W2295665692","doi":"10.5555/2034396.2034406","title":"Efficient planning in R-max","year":2011,"lang":"en","type":"article","venue":"Adaptive Agents and Multi-Agents Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Mathematical optimization; Markov decision process; Value (mathematics); Artificial intelligence; Algorithm; Machine learning; Mathematics; Markov process","score_opus":0.12420306409074311,"score_gpt":0.2922210493630056,"score_spread":0.16801798527226247,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2295665692","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012274354,0.0002920233,0.9777425,0.00026771636,0.000027019936,0.00009720672,0.00012481495,0.0009348306,0.008239472],"genre_scores_gemma":[0.39411822,0.00040032595,0.597923,0.0002227628,0.000036721067,0.0004050702,0.00037051187,0.00039161448,0.006131679],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984754,0.0006714217,0.00007578097,0.000387853,0.00021778981,0.00017165074],"domain_scores_gemma":[0.9972608,0.0019393293,0.00020961241,0.0003195007,0.00017789117,0.00009295864],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023183401,0.0009939711,0.001227201,0.00055135397,0.0006931092,0.0012954823,0.0016023392,0.0011522098,0.0064850673],"category_scores_gemma":[0.006433127,0.00067765807,0.00095378654,0.00081136206,0.0018835327,0.002297632,0.0019986674,0.0016081678,0.0012144956],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000194701,0.00007243508,0.00042442253,0.00021750918,0.000039474322,0.00011330791,0.00013761608,0.8238164,0.0013342566,0.10131669,0.003045019,0.069288164],"study_design_scores_gemma":[0.000032859938,0.000047731657,0.00008069953,0.000019185009,0.000011038015,0.00003192512,0.000028188468,0.90666336,0.0015490793,0.08909764,0.0024270418,0.000011249045],"about_ca_topic_score_codex":0.0032531188,"about_ca_topic_score_gemma":0.004750637,"teacher_disagreement_score":0.0064850673,"about_ca_system_score_codex":0.0013074073,"about_ca_system_score_gemma":0.0025886944,"threshold_uncertainty_score":0.02169472},"labels":[],"label_agreement":null},{"id":"W233883780","doi":"10.1609/icaps.v21i1.13468","title":"Distributed Control of Situated Assistance in Large Domains with Many Tasks","year":2011,"lang":"en","type":"article","venue":"Proceedings of the International Conference on Automated Planning and Scheduling","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Situated; Partially observable Markov decision process; Computer science; Task (project management); Scalability; Affordance; Domain (mathematical analysis); Set (abstract data type); Process (computing); Population; Markov decision process; Artificial intelligence; Human–computer interaction; Controller (irrigation); Control (management); Machine learning; Distributed computing; Markov process; Markov chain; Markov model; Engineering; Mathematics","score_opus":0.029838982046658497,"score_gpt":0.25990372975629306,"score_spread":0.23006474770963456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W233883780","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08842686,0.00018797014,0.90696764,0.00035028154,0.00006718203,0.00009802449,0.000042549593,0.00059238175,0.0032670766],"genre_scores_gemma":[0.9604593,0.000065005515,0.03770987,0.00005132174,0.000024312063,0.00014329031,0.000040346218,0.000027654414,0.001478948],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992625,0.00016808613,0.00003442586,0.0002618368,0.00013293308,0.0001401942],"domain_scores_gemma":[0.997658,0.0014361431,0.00027679888,0.00019165617,0.0002222553,0.00021517373],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014618336,0.0009725096,0.0013142732,0.00032980196,0.000943075,0.0011679033,0.0017224378,0.0010051822,0.0020824783],"category_scores_gemma":[0.004090379,0.0006910619,0.0006270625,0.00037149698,0.0019072444,0.0012316715,0.0028448326,0.0017031204,0.00025039152],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000079160156,0.00004435632,0.00028308952,0.0000321525,0.000019978017,0.00008403161,0.000073248615,0.9864071,0.0012787175,0.0030088532,0.00017778318,0.008511489],"study_design_scores_gemma":[0.00002788714,0.000026342801,0.00007781457,0.0000020453895,0.0000036919607,0.0000075585826,0.000013993772,0.9963407,0.00022431264,0.0031194817,0.00015269114,0.0000033603633],"about_ca_topic_score_codex":0.010991991,"about_ca_topic_score_gemma":0.0087357825,"teacher_disagreement_score":0.010991991,"about_ca_system_score_codex":0.0011744457,"about_ca_system_score_gemma":0.0016772465,"threshold_uncertainty_score":0.02185601},"labels":[],"label_agreement":null},{"id":"W2342981176","doi":"10.1007/978-3-319-57969-6_1","title":"NeuroHex: A Deep Q-learning Hex Agent","year":2017,"lang":"en","type":"preprint","venue":"Communications in computer and information science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Olympiad; Reinforcement learning; Computer science; Q-learning; Champion; Artificial intelligence; Initialization; State (computer science); Convolutional neural network; Action (physics); Algorithm; Mathematics; Political science; Law; Mathematics education","score_opus":0.05352251106668458,"score_gpt":0.32683744967227296,"score_spread":0.27331493860558836,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2342981176","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02915014,0.0002982219,0.9482116,0.0008462164,0.00037140798,0.00022814865,0.00040937954,0.00593418,0.014550742],"genre_scores_gemma":[0.5569311,0.0002411256,0.41677824,0.0007570624,0.000085267624,0.00036477763,0.00046707722,0.0003883489,0.023987005],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999821,0.000046401634,0.0000095329215,0.00003641189,0.000053040563,0.00003364246],"domain_scores_gemma":[0.99966514,0.000111834335,0.000021357315,0.00005601252,0.00007866652,0.00006708988],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005839162,0.00042421158,0.0006685969,0.0002419577,0.00046419862,0.00075424503,0.0014269095,0.0010802391,0.0126778595],"category_scores_gemma":[0.0017165927,0.00032499124,0.00031156372,0.00025614022,0.00065080595,0.00094622513,0.0018733838,0.0011667485,0.0018607157],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00093983207,0.00040043378,0.0021390498,0.0002278914,0.00012400081,0.0002751809,0.00012748285,0.56804705,0.0069301906,0.06476507,0.02306118,0.33296266],"study_design_scores_gemma":[0.000085438696,0.000080903774,0.000070833,0.000010461946,0.00001024821,0.000023020504,0.000013074003,0.98134285,0.0017035623,0.012091922,0.0045604175,0.000007269135],"about_ca_topic_score_codex":0.003770662,"about_ca_topic_score_gemma":0.00435163,"teacher_disagreement_score":0.0126778595,"about_ca_system_score_codex":0.00057739485,"about_ca_system_score_gemma":0.0013929582,"threshold_uncertainty_score":0.042411625},"labels":[],"label_agreement":null},{"id":"W2394540668","doi":"10.1609/aaai.v25i1.8032","title":"Provoking Opponents to Facilitate the Recognition of their Intentions","year":2011,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Defence Research and Development Canada; Université de Sherbrooke","funders":"Natural Sciences and Engineering Research Council of Canada; Fonds Québécois de la Recherche sur la Nature et les Technologies","keywords":"Adversary; Plan (archaeology); Contrast (vision); Observer (physics); Computer science; State (computer science); Psychology; Cognitive psychology; Artificial intelligence; Social psychology; Human–computer interaction; Computer security; Algorithm","score_opus":0.3161080705411999,"score_gpt":0.29375630951558374,"score_spread":0.022351761025616146,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2394540668","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6990337,0.00024686626,0.24379405,0.0016750059,0.00042716743,0.0003877393,0.00007977148,0.002552422,0.051803302],"genre_scores_gemma":[0.962234,0.0000534939,0.031222994,0.0003463845,0.000033991702,0.000091909154,0.00006297378,0.00008584066,0.005868444],"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993979,0.00019641855,0.00002495496,0.00014249267,0.00013518072,0.0001030312],"domain_scores_gemma":[0.99771035,0.0012936707,0.0002975949,0.0002883478,0.00016891242,0.00024120824],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010138955,0.00063663675,0.00024539474,0.00015560356,0.00026667767,0.0008037201,0.0005618173,0.00075618457,0.0074887383],"category_scores_gemma":[0.005787362,0.0003116861,0.00035647344,0.000055461205,0.00068737246,0.0008927088,0.0012868287,0.0013626256,0.0011306319],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019740688,0.0012883606,0.011744856,0.0004915057,0.00010793238,0.0014958632,0.0058428734,0.01328017,0.78423476,0.040018056,0.003570684,0.13595074],"study_design_scores_gemma":[0.0010808397,0.005537154,0.034568626,0.00023063454,0.00045525323,0.0020756566,0.0029147281,0.3377605,0.48688942,0.06975764,0.05842243,0.00030709174],"about_ca_topic_score_codex":0.00047012034,"about_ca_topic_score_gemma":0.0007708931,"teacher_disagreement_score":0.0074887383,"about_ca_system_score_codex":0.0002255877,"about_ca_system_score_gemma":0.00046188492,"threshold_uncertainty_score":0.025052309},"labels":[],"label_agreement":null},{"id":"W2395526823","doi":"","title":"An empirical analysis of reinforcement learning using design of experiments","year":2013,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Convergence (economics); Computer science; Reinforcement learning; Artificial neural network; Subspace topology; Design of experiments; Logistic regression; Regression; Artificial intelligence; Machine learning; Mathematical optimization; Algorithm; Statistics; Mathematics","score_opus":0.08287195625593569,"score_gpt":0.3503770929916817,"score_spread":0.267505136735746,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2395526823","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.62177455,0.000392724,0.37165573,0.00031100362,0.00019821517,0.002369466,0.00029079223,0.00026866738,0.0027388814],"genre_scores_gemma":[0.932436,0.00007225708,0.063728936,0.000116548916,0.000034363296,0.0029659793,0.00014092699,0.000038652866,0.00046629258],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.92948943,0.059803408,0.0023052003,0.0030996422,0.004291806,0.0010105562],"domain_scores_gemma":[0.3903629,0.5649916,0.018353447,0.01813478,0.0072247316,0.0009324943],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0857745,0.0011452487,0.001270743,0.0010666522,0.0004487301,0.0014955022,0.0015055026,0.0014375856,0.0012896844],"category_scores_gemma":[0.2300372,0.00062231295,0.0011972848,0.00062006316,0.0024211383,0.0014693674,0.0009734436,0.0019422328,0.00014588998],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.030650005,0.01463484,0.096133515,0.0030675402,0.005058466,0.000541796,0.0028765253,0.51658744,0.02615046,0.11338759,0.0039923727,0.18691945],"study_design_scores_gemma":[0.0026893958,0.029651351,0.018671993,0.00022366791,0.00068078353,0.00018219015,0.0003294226,0.9035986,0.01592159,0.024893101,0.00301341,0.00014453623],"about_ca_topic_score_codex":0.00063200435,"about_ca_topic_score_gemma":0.0003924114,"teacher_disagreement_score":0.0857745,"about_ca_system_score_codex":0.0022607916,"about_ca_system_score_gemma":0.0012770595,"threshold_uncertainty_score":0.45362437},"labels":[],"label_agreement":null},{"id":"W2396459401","doi":"","title":"SmartWheeler: A Robotic Wheelchair Test-Bed for Investigating New Models of Human-Robot Interaction.","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Wheelchair; Robot; Human–computer interaction; Computer science; Human–robot interaction; Mobile robot; Autonomy; Scale (ratio); Robotics; Interface (matter); Control (management); Test (biology); Artificial intelligence; Simulation; Engineering","score_opus":0.0721783322261814,"score_gpt":0.31582696792387427,"score_spread":0.24364863569769285,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2396459401","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46535933,0.0007372242,0.5119689,0.00062839396,0.00022264653,0.0010128532,0.0032371439,0.006772153,0.010061315],"genre_scores_gemma":[0.7996738,0.00054620916,0.18342704,0.00011356184,0.00001956997,0.0014754934,0.003974704,0.00025967238,0.010509988],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9997861,0.00008932303,0.00000939283,0.000035926,0.000057566936,0.000021652984],"domain_scores_gemma":[0.9995896,0.00021251668,0.000031381685,0.000059814254,0.000047509144,0.000059139842],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005156693,0.00086779613,0.0005832337,0.0003202876,0.0003651428,0.00057792646,0.0016020135,0.0011897394,0.005305135],"category_scores_gemma":[0.00133656,0.0003415535,0.00044392375,0.00022480142,0.0006294367,0.0009496752,0.00084675004,0.00053307466,0.0011598809],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022694673,0.0024709527,0.0039669867,0.0013099248,0.0003895346,0.00077097927,0.00058442674,0.7513201,0.1104197,0.02652764,0.016673656,0.08329662],"study_design_scores_gemma":[0.0001548052,0.00096316874,0.0011590377,0.000037174126,0.00004668664,0.00012878,0.000085251944,0.9703664,0.014103336,0.00493834,0.00797289,0.000044081462],"about_ca_topic_score_codex":0.0044207363,"about_ca_topic_score_gemma":0.0041918526,"teacher_disagreement_score":0.005305135,"about_ca_system_score_codex":0.00037362552,"about_ca_system_score_gemma":0.00065658306,"threshold_uncertainty_score":0.017747462},"labels":[],"label_agreement":null},{"id":"W2401626140","doi":"","title":"Treating Epilepsy by Reinforcement Learning Via Manifold-Based Simulation.","year":2010,"lang":"en","type":"article","venue":"National Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Epilepsy; Artificial intelligence; Manifold (fluid mechanics); Psychology; Neuroscience; Engineering; Mechanical engineering","score_opus":0.06475144094737749,"score_gpt":0.32914766654567496,"score_spread":0.26439622559829745,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2401626140","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.074284755,0.0017577072,0.9107772,0.0019401865,0.00021443787,0.00016124806,0.000082359096,0.00080387446,0.009978224],"genre_scores_gemma":[0.9645454,0.0005072891,0.033303447,0.000104242485,0.000027314449,0.00012324034,0.00003663694,0.000032436496,0.001320051],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9999268,0.000030116325,0.0000045224424,0.000010070969,0.000015949816,0.000012503452],"domain_scores_gemma":[0.9997441,0.00015589404,0.000030420875,0.000017849346,0.000023131506,0.000028597478],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00027643002,0.0003910838,0.00044329092,0.00020009586,0.0002041348,0.0003887068,0.0005292804,0.0005292696,0.0017851569],"category_scores_gemma":[0.0013539747,0.00014661795,0.0003517143,0.000103465936,0.000397714,0.00038321118,0.0007662417,0.00069840706,0.0002246163],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019930064,0.000099965146,0.00085229834,0.00007188405,0.000056839857,0.00022182424,0.0000552573,0.9406993,0.00387323,0.0063465014,0.001973646,0.045549914],"study_design_scores_gemma":[0.000039280232,0.00008465795,0.00012340023,0.000010339734,0.000013431328,0.00008234633,0.000010340691,0.99165195,0.00084252266,0.0063603283,0.000775597,0.0000058176647],"about_ca_topic_score_codex":0.001923398,"about_ca_topic_score_gemma":0.0019974036,"teacher_disagreement_score":0.001923398,"about_ca_system_score_codex":0.00029319583,"about_ca_system_score_gemma":0.00055576593,"threshold_uncertainty_score":0.005971968},"labels":[],"label_agreement":null},{"id":"W2403918565","doi":"","title":"Online Robot Task Switching Under Diminishing Returns.","year":2010,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Foraging; Computer science; Task (project management); Robot; Variety (cybernetics); Work (physics); Artificial intelligence; Mathematical optimization; Mathematics; Ecology; Engineering","score_opus":0.01661103039644213,"score_gpt":0.260694048774749,"score_spread":0.24408301837830687,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2403918565","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23831995,0.0005154697,0.75360984,0.0005957863,0.00006396351,0.00012903044,0.00010643392,0.0007320023,0.0059275627],"genre_scores_gemma":[0.96011543,0.00008061552,0.037240453,0.000065819506,0.000020049534,0.000094668605,0.000053308053,0.00004908037,0.0022806476],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999268,0.0003156656,0.000029550809,0.00011498902,0.0001457872,0.00012596525],"domain_scores_gemma":[0.994234,0.004003619,0.00061456475,0.0004980306,0.00030561362,0.000344094],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019371179,0.0006486131,0.00095505285,0.0004939097,0.00036321903,0.0006042649,0.0017009078,0.0009484699,0.0026585006],"category_scores_gemma":[0.011135174,0.00033142965,0.00044255352,0.000367023,0.0010261739,0.0013951196,0.0011380921,0.0010577847,0.00042014004],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004584908,0.00027813963,0.0021455132,0.000099542915,0.00007447496,0.00016063688,0.0001536798,0.8741711,0.0038665673,0.022989811,0.0017254667,0.09387667],"study_design_scores_gemma":[0.000021173855,0.000056061246,0.00024161165,0.0000028911338,0.0000059891913,0.000023531205,0.000010900635,0.9873611,0.0006619326,0.011358138,0.00025131882,0.000005323982],"about_ca_topic_score_codex":0.0022784478,"about_ca_topic_score_gemma":0.0017175651,"teacher_disagreement_score":0.0026585006,"about_ca_system_score_codex":0.0010082782,"about_ca_system_score_gemma":0.00067617366,"threshold_uncertainty_score":0.010244608},"labels":[],"label_agreement":null},{"id":"W2406813038","doi":"","title":"Learning and planning with timing information in Markov decision processes","year":2015,"lang":"en","type":"article","venue":"Uncertainty in Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Markov decision process; Representation (politics); Artificial intelligence; Machine learning; Markov chain; Markov process; Set (abstract data type); State (computer science); Robot; Duration (music); Algorithm; Mathematics","score_opus":0.05173460047726749,"score_gpt":0.3098116824769045,"score_spread":0.258077081999637,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2406813038","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08509644,0.00024914616,0.9126031,0.0004296591,0.000026438323,0.000028665514,0.00007334019,0.00018101843,0.0013123002],"genre_scores_gemma":[0.9035395,0.00025675818,0.09464844,0.000077256766,0.000037898095,0.000102745245,0.00014048137,0.000046967405,0.0011499593],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99886155,0.0005912927,0.00005668914,0.00021656792,0.00015651851,0.000117491865],"domain_scores_gemma":[0.99332607,0.005645429,0.00042124675,0.00022023842,0.00021300686,0.00017403213],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023630857,0.0007817065,0.0011151716,0.00056273997,0.00053313765,0.00111546,0.0012041648,0.0013053456,0.0016284814],"category_scores_gemma":[0.009541492,0.0006235167,0.0008021451,0.00081385847,0.0020510978,0.002978705,0.0013396383,0.001574499,0.00017692994],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008156322,0.000020958518,0.00028851113,0.000027086433,0.000015048909,0.000048957936,0.000044905253,0.95543516,0.000341535,0.036147196,0.00010474743,0.0074442644],"study_design_scores_gemma":[0.000011964532,0.000011706159,0.0000320744,0.0000028042637,0.0000028970464,0.0000044532953,0.00000383819,0.9674666,0.00017384096,0.032221995,0.00006387462,0.000004009566],"about_ca_topic_score_codex":0.0062187226,"about_ca_topic_score_gemma":0.004011709,"teacher_disagreement_score":0.0062187226,"about_ca_system_score_codex":0.0013768327,"about_ca_system_score_gemma":0.0012519748,"threshold_uncertainty_score":0.0124973655},"labels":[],"label_agreement":null},{"id":"W2407045295","doi":"10.13140/2.1.3356.8002","title":"Efficient Abstraction Selection in Reinforcement Learning --- Extended Abstract","year":2013,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Abstraction; Reinforcement learning; Computer science; Markov decision process; Selection (genetic algorithm); Set (abstract data type); Artificial intelligence; State (computer science); Markov process; Machine learning; Theoretical computer science; Programming language; Mathematics","score_opus":0.012565001163403085,"score_gpt":0.24342780531320649,"score_spread":0.2308628041498034,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2407045295","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027380213,0.0002025311,0.9701531,0.00021392487,0.00004096353,0.000053937954,0.00004812536,0.00036610357,0.0015411376],"genre_scores_gemma":[0.8698692,0.00014279128,0.12769623,0.00009384981,0.0000353763,0.00014079337,0.0001043578,0.000068216665,0.0018491811],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9988525,0.00052896555,0.000056833484,0.00018307184,0.00023800568,0.00014058375],"domain_scores_gemma":[0.9980026,0.0012326547,0.00016904747,0.00025267468,0.00022562966,0.00011734986],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001828485,0.00058974593,0.001260691,0.00030164528,0.00033763997,0.0008094029,0.0010976745,0.00066434284,0.0024403993],"category_scores_gemma":[0.0050396007,0.0003470342,0.00055549026,0.0004137789,0.0009209566,0.0013969683,0.0017575301,0.0013708977,0.00032230743],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022957043,0.00007457939,0.0009966837,0.00009822497,0.000055619978,0.00013784548,0.00008021477,0.8830867,0.0024918234,0.050226912,0.0012594734,0.06126243],"study_design_scores_gemma":[0.000015821235,0.000026174494,0.000052418003,0.0000053036874,0.000004927497,0.000009564388,0.0000036415431,0.98119605,0.00038631543,0.018004997,0.0002915272,0.0000032719192],"about_ca_topic_score_codex":0.002763004,"about_ca_topic_score_gemma":0.0016821272,"teacher_disagreement_score":0.002763004,"about_ca_system_score_codex":0.0008972128,"about_ca_system_score_gemma":0.0010094831,"threshold_uncertainty_score":0.009670079},"labels":[],"label_agreement":null},{"id":"W2407441816","doi":"","title":"Reinforcement Learning with Limited Reinforcement: Using Bayes Risk for Active Learning in POMDPs.","year":2008,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Partially observable Markov decision process; Reinforcement learning; Computer science; Planner; Markov decision process; Artificial intelligence; Automated planning and scheduling; Machine learning; Bayesian probability; Domain (mathematical analysis); Markov process; Markov chain; Markov model; Mathematics","score_opus":0.029598965353824374,"score_gpt":0.2529070386760443,"score_spread":0.22330807332221994,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2407441816","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010092512,0.0005847075,0.987089,0.00034132306,0.000044048113,0.0000623784,0.00003689248,0.00025695452,0.0014920994],"genre_scores_gemma":[0.7824432,0.0005087792,0.21405913,0.00024521438,0.0001082482,0.00042808606,0.00013410281,0.00012219904,0.0019510654],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997575,0.0014560121,0.0001044676,0.0002903835,0.00041383237,0.00016039956],"domain_scores_gemma":[0.98637366,0.011541617,0.0006854648,0.00041156687,0.0005886202,0.00039904207],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005747107,0.0014037357,0.0022810402,0.00080616574,0.00061583536,0.0015238218,0.002423354,0.0018222053,0.0024925962],"category_scores_gemma":[0.02146722,0.0010006059,0.00086343463,0.00066338433,0.0022756672,0.0033794139,0.0024960202,0.0029480192,0.0003561621],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015181857,0.00008286697,0.0011449932,0.0001079348,0.00006989268,0.00007344058,0.00011247392,0.92657524,0.00023964341,0.03523748,0.0007359002,0.035468403],"study_design_scores_gemma":[0.00001300478,0.0000140133,0.000027629701,0.000008779022,0.0000053972058,0.0000042700176,0.000003508815,0.9853579,0.00006578336,0.014387782,0.000108709224,0.0000031935058],"about_ca_topic_score_codex":0.00689328,"about_ca_topic_score_gemma":0.0051845727,"teacher_disagreement_score":0.00689328,"about_ca_system_score_codex":0.0020403352,"about_ca_system_score_gemma":0.0019233293,"threshold_uncertainty_score":0.030393958},"labels":[],"label_agreement":null},{"id":"W2424764846","doi":"10.1109/syscon.2016.7490540","title":"A learning invader for the guarding a territory game","year":2016,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Royal Military College of Canada; Carleton University","funders":"","keywords":"Guard (computer science); Nash equilibrium; Computer science; Robot; Game theory; Artificial intelligence; Mathematical economics; Mathematics","score_opus":0.021743686676913006,"score_gpt":0.24524124685625984,"score_spread":0.22349756017934683,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2424764846","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16618243,0.00011688535,0.8168479,0.0006412141,0.000046875244,0.0001985759,0.000036347927,0.00033099347,0.015598733],"genre_scores_gemma":[0.90776426,0.000094506206,0.08406464,0.00012121304,0.00001939736,0.00016238609,0.000035837886,0.000032426586,0.0077052326],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993919,0.00027692632,0.000022796214,0.000095639865,0.00012723477,0.000085502885],"domain_scores_gemma":[0.9990577,0.0004916429,0.000112850495,0.000052809304,0.00009201897,0.0001928908],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010359051,0.0005246971,0.00069092034,0.00032561054,0.00058374094,0.00092189206,0.0013624785,0.0011918986,0.0033590742],"category_scores_gemma":[0.002720085,0.00017047228,0.00045283095,0.00017729269,0.0015230791,0.0014503221,0.001458732,0.001394753,0.00029955266],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034841162,0.00039452358,0.0032927373,0.00016962358,0.00009871954,0.00061532337,0.0006926145,0.76747495,0.008179543,0.17205103,0.0012525563,0.045429897],"study_design_scores_gemma":[0.000037751313,0.00015131563,0.00014073975,0.000008194701,0.0000094723,0.00007208941,0.000056930578,0.983076,0.00077249645,0.014378466,0.0012840068,0.000012505965],"about_ca_topic_score_codex":0.002319059,"about_ca_topic_score_gemma":0.0019018378,"teacher_disagreement_score":0.0033590742,"about_ca_system_score_codex":0.0007555359,"about_ca_system_score_gemma":0.0010342465,"threshold_uncertainty_score":0.011237264},"labels":[],"label_agreement":null},{"id":"W2431563754","doi":"","title":"On the influence of learning time on evolutionary online learning of cooperative behavior","year":2001,"lang":"en","type":"article","venue":"Genetic and Evolutionary Computation Conference","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Offline learning; Computer science; Reinforcement learning; Artificial intelligence; Machine learning; Key (lock); Online learning; Action (physics); Evolutionary computation; Duration (music); Evolutionary algorithm","score_opus":0.020548650184544874,"score_gpt":0.2539995995415321,"score_spread":0.23345094935698724,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2431563754","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32548684,0.0007827198,0.66435516,0.0007304286,0.00006863119,0.000060728875,0.000029970197,0.00024498889,0.008240615],"genre_scores_gemma":[0.9736482,0.00022043155,0.024954561,0.000071375456,0.000028504162,0.00006043673,0.000014760361,0.00004810158,0.0009535975],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998936,0.0005765927,0.000044724446,0.00012837774,0.00018838282,0.00012584752],"domain_scores_gemma":[0.95895976,0.03721003,0.0012714932,0.0010966717,0.0008128504,0.0006491594],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029462718,0.0005864953,0.0007291923,0.00031123954,0.0004439796,0.00089552137,0.000795352,0.0007516261,0.0022034706],"category_scores_gemma":[0.033185765,0.00023070988,0.00031917103,0.00024596113,0.0013535634,0.0018427111,0.0011399155,0.0014117546,0.00018901342],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048837997,0.00037003707,0.0027856368,0.00013883087,0.00007609452,0.00020063575,0.00025255568,0.8587828,0.008907528,0.037218306,0.00041890162,0.09036029],"study_design_scores_gemma":[0.00003161181,0.0002478202,0.00045802552,0.000009082815,0.000016978742,0.000038977418,0.000024584202,0.98343116,0.0021468997,0.013349621,0.00023342489,0.0000118164135],"about_ca_topic_score_codex":0.00091904524,"about_ca_topic_score_gemma":0.0007546142,"teacher_disagreement_score":0.0029462718,"about_ca_system_score_codex":0.0005234246,"about_ca_system_score_gemma":0.0005014507,"threshold_uncertainty_score":0.015581548},"labels":[],"label_agreement":null},{"id":"W246639213","doi":"","title":"Prediction Driven Behavior: Learning Predictions that Drive Fixed Responses","year":2014,"lang":"en","type":"article","venue":"National Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Robot; Computer science; Artificial intelligence; Adaptive behavior; Behavior-based robotics; Adaptive control; Reinforcement learning; Latency (audio); Control (management); Control theory (sociology); Simulation; Mobile robot; Psychology","score_opus":0.15524353064616092,"score_gpt":0.3359340618331985,"score_spread":0.1806905311870376,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W246639213","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03189036,0.000081613936,0.9645056,0.0001259042,0.000056724253,0.000051164443,0.000027790631,0.0016053383,0.0016554997],"genre_scores_gemma":[0.7685532,0.00009983656,0.22741008,0.00016476648,0.00004104598,0.00017438346,0.00007776422,0.00017350905,0.0033053844],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99959546,0.00008730506,0.000017662507,0.00013070593,0.00013005725,0.000038859165],"domain_scores_gemma":[0.99882776,0.00053932983,0.0001903991,0.00022219103,0.00015023348,0.000070038506],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00076316966,0.0006059833,0.0003060449,0.00017938475,0.00019085412,0.0004224298,0.0014026493,0.00054517196,0.0017260071],"category_scores_gemma":[0.002574152,0.00028860508,0.00032457354,0.0001367124,0.0007866278,0.0006326709,0.00056576,0.0009106494,0.0003826868],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042672033,0.0004202014,0.0047517754,0.00017996762,0.0001209688,0.00019895624,0.00027981814,0.43364763,0.10897233,0.022636756,0.0032632123,0.42510164],"study_design_scores_gemma":[0.000026542222,0.00014222068,0.00057332026,0.0000084614585,0.000012669199,0.000044843746,0.000008298226,0.97885305,0.013735438,0.0051590144,0.0014178435,0.000018233992],"about_ca_topic_score_codex":0.0016600041,"about_ca_topic_score_gemma":0.0015115285,"teacher_disagreement_score":0.0017260071,"about_ca_system_score_codex":0.00034440734,"about_ca_system_score_gemma":0.00054361677,"threshold_uncertainty_score":0.005774021},"labels":[],"label_agreement":null},{"id":"W2485417938","doi":"10.4018/978-1-59904-552-8.ch009","title":"Monocular Vision System that Learns with Approximation Spaces","year":2011,"lang":"en","type":"book-chapter","venue":"IGI Global eBooks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Artificial intelligence; Reinforcement learning; Computer science; Computer vision; Machine vision; Robot; Crawling","score_opus":0.023229796509762136,"score_gpt":0.22607563529727262,"score_spread":0.20284583878751047,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2485417938","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011828165,0.00040399248,0.9816046,0.00012838605,0.00006265652,0.00003632321,0.000055054876,0.0011833535,0.004697568],"genre_scores_gemma":[0.5497804,0.0005396407,0.436739,0.00033339838,0.00007537415,0.0001382497,0.00022346877,0.000112972746,0.012057477],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99965715,0.000044893877,0.000012233548,0.00012929467,0.000113672046,0.000042752046],"domain_scores_gemma":[0.99967945,0.0000760207,0.00004703189,0.00008350203,0.00008260409,0.000031388157],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00050558185,0.00049640064,0.00063593994,0.00023656985,0.00029883295,0.0007453198,0.00095881463,0.0007837278,0.0027212712],"category_scores_gemma":[0.0014572102,0.0002796068,0.0004450387,0.00026542027,0.00050454296,0.0009845697,0.0010135102,0.0008727167,0.00083348603],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028891047,0.0001507763,0.0017095258,0.00016707394,0.00013619463,0.00015462264,0.00021294382,0.35398832,0.049349524,0.056256898,0.0058734166,0.5317117],"study_design_scores_gemma":[0.000018008108,0.00008949837,0.0005146755,0.000008833636,0.000011257199,0.000064915526,0.000011850181,0.9790678,0.00568613,0.009232871,0.0052791503,0.000015003967],"about_ca_topic_score_codex":0.004485505,"about_ca_topic_score_gemma":0.004344031,"teacher_disagreement_score":0.004485505,"about_ca_system_score_codex":0.0007053911,"about_ca_system_score_gemma":0.00087239296,"threshold_uncertainty_score":0.009103596},"labels":[],"label_agreement":null},{"id":"W2490120274","doi":"10.4018/978-1-60960-165-2.ch003","title":"An Introduction to Fully and Partially Observable Markov Decision Processes","year":2011,"lang":"en","type":"book-chapter","venue":"IGI Global eBooks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Management science; Partially observable Markov decision process; Markov decision process; Focus (optics); Markov process; Software deployment; Markov chain; Observable; Markov model; Risk analysis (engineering); Machine learning; Mathematics; Software engineering; Engineering; Business","score_opus":0.02260650668927809,"score_gpt":0.24811897239786734,"score_spread":0.22551246570858924,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2490120274","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0022595502,0.12733255,0.5531158,0.006394117,0.002786585,0.00020770304,0.0021301084,0.0014032865,0.30437043],"genre_scores_gemma":[0.093539804,0.23678444,0.3687333,0.004959445,0.005611997,0.00090739684,0.0039662127,0.000853307,0.2846441],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9996942,0.00006454165,0.000026625767,0.000071180766,0.00011599252,0.000027540842],"domain_scores_gemma":[0.9992908,0.00054189813,0.00002786027,0.00004222346,0.000069300215,0.000027857108],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00049531186,0.0011525056,0.00074983254,0.00079918257,0.0003663272,0.0014405964,0.0008536616,0.0014378409,0.036572717],"category_scores_gemma":[0.0012810132,0.00058520626,0.0007409175,0.0019182666,0.0008444556,0.0023957642,0.00082288415,0.002771585,0.011222042],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000027367978,0.00008219884,0.00019272392,0.0012271858,0.000032202457,0.00024849887,0.00028556157,0.014441754,0.0012091433,0.6546868,0.10392044,0.22364616],"study_design_scores_gemma":[0.000008789619,0.000032734162,0.0002807159,0.00044292005,0.000010034145,0.00029950382,0.00003294116,0.009467384,0.00025597308,0.37539786,0.61374813,0.000022913779],"about_ca_topic_score_codex":0.0011747422,"about_ca_topic_score_gemma":0.0013472845,"teacher_disagreement_score":0.036572717,"about_ca_system_score_codex":0.0010769556,"about_ca_system_score_gemma":0.0009865161,"threshold_uncertainty_score":0.12234795},"labels":[],"label_agreement":null},{"id":"W2496219042","doi":"10.4018/978-1-60960-171-3.ch007","title":"Reward Shaping and Mixed Resolution Function Approximation","year":2011,"lang":"en","type":"book-chapter","venue":"IGI Global eBooks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Representation (politics); Function approximation; Function (biology); Reinforcement learning; Computer science; Space (punctuation); Core (optical fiber); Resolution (logic); Function space; Artificial intelligence; Quality (philosophy); Process (computing); Algorithm; Mathematics; Artificial neural network; Discrete mathematics","score_opus":0.043424739399203244,"score_gpt":0.23376106207356487,"score_spread":0.19033632267436162,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2496219042","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0031947193,0.00077231554,0.9886098,0.00019033317,0.000038428032,0.000025913776,0.000013838683,0.00023766136,0.0069170827],"genre_scores_gemma":[0.3828178,0.0019427246,0.5895984,0.00032193356,0.000113996924,0.000296977,0.00010277209,0.00029336772,0.024512034],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999198,0.00032287338,0.000034575445,0.00013026541,0.0002422956,0.00007203262],"domain_scores_gemma":[0.9986986,0.00089811656,0.00009354261,0.00013348661,0.000118486445,0.000057697074],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014891125,0.0009990908,0.0009426008,0.0005219852,0.00031979117,0.001726828,0.0015446969,0.0016772338,0.006046676],"category_scores_gemma":[0.004967592,0.00051170815,0.00088749244,0.00065782166,0.0014390047,0.0018623583,0.0016231887,0.002620343,0.0013000062],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013456536,0.00007742043,0.0003107874,0.0002529724,0.000048373397,0.00013250274,0.00017884486,0.5110839,0.0045759967,0.30100057,0.0030903416,0.17911366],"study_design_scores_gemma":[0.000011981581,0.000050385057,0.00006568168,0.000046091656,0.000010895442,0.00006035221,0.000011076439,0.929436,0.0012994043,0.06476769,0.004227578,0.000012856861],"about_ca_topic_score_codex":0.0009212453,"about_ca_topic_score_gemma":0.0006837809,"teacher_disagreement_score":0.006046676,"about_ca_system_score_codex":0.0013904461,"about_ca_system_score_gemma":0.0007392453,"threshold_uncertainty_score":0.020228148},"labels":[],"label_agreement":null},{"id":"W2510392992","doi":"10.1007/978-3-319-46073-4_6","title":"A Neural Field Approach to Obstacle Avoidance","year":2016,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Artificial intelligence; Computer science; Robotics; Flexibility (engineering); Obstacle avoidance; Obstacle; Field (mathematics); Task (project management); Focus (optics); Cognitive robotics; Human–computer interaction; Robot; Mobile robot; Engineering; Systems engineering","score_opus":0.01967386230579018,"score_gpt":0.24170938995895433,"score_spread":0.22203552765316414,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2510392992","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0075285984,0.005261561,0.91624665,0.00090706424,0.0006165604,0.000028809753,0.000088456305,0.00023610245,0.06908621],"genre_scores_gemma":[0.5339482,0.009733284,0.29997668,0.00070876844,0.0010907307,0.00020823989,0.00026828874,0.00025010243,0.15381579],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9999479,0.000009373142,0.0000021070239,0.000011900889,0.000021634625,0.0000071205986],"domain_scores_gemma":[0.9999362,0.000029181278,0.000005700765,0.0000049709306,0.000015974536,0.000007901787],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00012720801,0.00054740626,0.0005076385,0.00052322855,0.0003206145,0.0005687805,0.0014326253,0.0011724142,0.0072228345],"category_scores_gemma":[0.0003889754,0.00025789058,0.0004598892,0.00057145214,0.00061287434,0.0010786164,0.00069097464,0.0013599887,0.0007580691],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000031545827,0.00005778322,0.00012607555,0.0001407678,0.000036539543,0.00006935931,0.000051429794,0.3889869,0.002726718,0.464304,0.010229446,0.13323958],"study_design_scores_gemma":[0.000009445766,0.000023975941,0.00011285525,0.000026261398,0.000007911404,0.000043047603,0.000011836262,0.6409141,0.00037392756,0.3495468,0.008916252,0.00001360005],"about_ca_topic_score_codex":0.004023273,"about_ca_topic_score_gemma":0.0033374322,"teacher_disagreement_score":0.0072228345,"about_ca_system_score_codex":0.00064852746,"about_ca_system_score_gemma":0.00038519801,"threshold_uncertainty_score":0.02416277},"labels":[],"label_agreement":null},{"id":"W2519270595","doi":"","title":"Multi-step linear Dyna-style planning","year":2009,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Iterated function; Projection (relational algebra); Linear model; Feature (linguistics); Algorithm; Mathematics; Machine learning","score_opus":0.035637737076629206,"score_gpt":0.3005928994693765,"score_spread":0.2649551623927473,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2519270595","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0063331593,0.0000991416,0.9888809,0.00015487081,0.000037649545,0.00006337231,0.00006835466,0.0009361423,0.0034264417],"genre_scores_gemma":[0.5196259,0.00017873751,0.47073713,0.00019612286,0.00003009917,0.00034798158,0.0002048454,0.00018712292,0.008492108],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995382,0.00012432561,0.000028571661,0.0001225565,0.0001298598,0.000056533438],"domain_scores_gemma":[0.99899226,0.0006171516,0.00008686067,0.000109917746,0.00012954081,0.000064245985],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008062184,0.0008033824,0.0006611988,0.00029258485,0.0004452214,0.0007983818,0.0014194152,0.0007900497,0.0064388374],"category_scores_gemma":[0.0021820674,0.00049973483,0.0005900085,0.00033436654,0.00084477,0.0009154773,0.001320777,0.0012290657,0.0009294075],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000080643054,0.00004506816,0.00044815094,0.00009766069,0.0000283326,0.00009298616,0.00008951035,0.9157818,0.0011842864,0.029813984,0.0014000372,0.050937545],"study_design_scores_gemma":[0.0000104197525,0.000017674412,0.000023048526,0.000004398196,0.000003519033,0.000012323993,0.000005050772,0.9919419,0.0004559139,0.006567975,0.00095363724,0.0000041043027],"about_ca_topic_score_codex":0.0051573277,"about_ca_topic_score_gemma":0.0071673375,"teacher_disagreement_score":0.0064388374,"about_ca_system_score_codex":0.0008354361,"about_ca_system_score_gemma":0.0018094452,"threshold_uncertainty_score":0.021540105},"labels":[],"label_agreement":null},{"id":"W2520501711","doi":"","title":"Regularized policy iteration with nonparametric function spaces","year":2016,"lang":"en","type":"article","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":59,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Markov decision process; Mathematical optimization; Reproducing kernel Hilbert space; Reinforcement learning; Mathematics; Regularization (linguistics); Function space; Nonparametric statistics; Bellman equation; Minimax; Function (biology); Computer science; Applied mathematics; Hilbert space; Markov process; Artificial intelligence; Discrete mathematics","score_opus":0.008502984474690533,"score_gpt":0.21923521494304662,"score_spread":0.2107322304683561,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2520501711","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01164427,0.000121347366,0.98731905,0.00014330265,0.000018116942,0.000019409039,0.000011576875,0.00010640081,0.0006166766],"genre_scores_gemma":[0.6549669,0.00019404358,0.34158558,0.00021953571,0.000053205582,0.00025381002,0.00009956912,0.00011074526,0.002516699],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99845314,0.0008075041,0.00006056154,0.00020384102,0.0003404117,0.0001345354],"domain_scores_gemma":[0.99176335,0.006250172,0.00061832165,0.00054180383,0.00061414676,0.00021214875],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003780708,0.00087402656,0.001424952,0.0004980514,0.0003932761,0.0010004982,0.001306849,0.0013635767,0.001227991],"category_scores_gemma":[0.015165364,0.0005250463,0.0006662961,0.0004851047,0.0019557443,0.0016253934,0.0016970211,0.0019607563,0.00027365435],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007152772,0.00003535136,0.0003247665,0.000041258492,0.000026380087,0.00002635567,0.000033580207,0.95524555,0.00059812004,0.029793799,0.0002472105,0.013556152],"study_design_scores_gemma":[0.00000356865,0.000011262133,0.000012804409,0.000001977864,9.834043e-7,0.0000030968442,0.0000011610331,0.9952909,0.0001327027,0.0044758385,0.00006408222,0.0000016138284],"about_ca_topic_score_codex":0.0026618745,"about_ca_topic_score_gemma":0.0018442018,"teacher_disagreement_score":0.003780708,"about_ca_system_score_codex":0.0014653752,"about_ca_system_score_gemma":0.0019377262,"threshold_uncertainty_score":0.019994557},"labels":[],"label_agreement":null},{"id":"W2523728418","doi":"10.1609/aaai.v31i1.10916","title":"The Option-Critic Architecture","year":2017,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":209,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Flexibility (engineering); Computer science; Architecture; Abstraction; Key (lock); Artificial intelligence; Economics; Computer security; Management; Epistemology","score_opus":0.08507794004948784,"score_gpt":0.3192055271439572,"score_spread":0.2341275870944694,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2523728418","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010012724,0.0004054291,0.97541124,0.0010000651,0.000096785356,0.00004522463,0.00008872127,0.0005385211,0.0124011915],"genre_scores_gemma":[0.7117746,0.00076225464,0.27234036,0.0004591023,0.00013311463,0.00028338487,0.00018099016,0.0001588672,0.0139074],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994487,0.00016181303,0.0000298946,0.00014483152,0.0001618101,0.000052911204],"domain_scores_gemma":[0.9990978,0.00043156222,0.00007432087,0.00015565782,0.00014343111,0.00009728146],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009809321,0.0008233829,0.0006473402,0.00034592763,0.00046799984,0.0012915645,0.0015277191,0.001421225,0.0046299663],"category_scores_gemma":[0.0040002763,0.00060600584,0.00060589786,0.00035267018,0.0016972633,0.0024243686,0.0018483688,0.003288094,0.0009195589],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008746639,0.000036731133,0.0006332994,0.00010139992,0.00006939502,0.00016621176,0.00011416661,0.5939296,0.003040292,0.34730422,0.0030457643,0.051471446],"study_design_scores_gemma":[0.000017970502,0.000021988417,0.00009102263,0.000016846216,0.0000143179095,0.000040283798,0.0000059849226,0.8564832,0.00053880387,0.14000203,0.0027545537,0.000013051197],"about_ca_topic_score_codex":0.0026436625,"about_ca_topic_score_gemma":0.0032571694,"teacher_disagreement_score":0.0046299663,"about_ca_system_score_codex":0.00093433383,"about_ca_system_score_gemma":0.0014179953,"threshold_uncertainty_score":0.015488803},"labels":[],"label_agreement":null},{"id":"W2538613533","doi":"10.1109/gem.2014.7048106","title":"Enabling motivated believable agents with reinforcement learning","year":2014,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Reinforcement learning; Computer science; Action (physics); Narrative; Reinforcement; Human–computer interaction; Video game; Variety (cybernetics); Artificial intelligence; Multimedia; Psychology; Social psychology","score_opus":0.014877786932652947,"score_gpt":0.22514617005840445,"score_spread":0.2102683831257515,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2538613533","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09361356,0.00017612363,0.8975568,0.00044894998,0.00004549758,0.00018395427,0.000027204662,0.0011331511,0.006814794],"genre_scores_gemma":[0.85743487,0.0001302342,0.13958266,0.00015191907,0.000015366566,0.0002529604,0.000041492738,0.00007160774,0.0023188633],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99938977,0.0002767288,0.000030219488,0.000079410805,0.00015237533,0.00007148025],"domain_scores_gemma":[0.9981668,0.0011583033,0.00022342024,0.00015433008,0.00016725728,0.00012982293],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012013928,0.0007642982,0.00035924846,0.0002419035,0.00030469027,0.00077668997,0.0010427858,0.0007612689,0.0018392518],"category_scores_gemma":[0.005142406,0.00038136553,0.00039882338,0.00011622649,0.0010804778,0.0010679017,0.0017208823,0.0013984871,0.00032740444],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017117466,0.0003873941,0.0022706722,0.00017659873,0.00006766283,0.00025262218,0.00045127427,0.85098886,0.015954293,0.056696683,0.0012090473,0.07137369],"study_design_scores_gemma":[0.00003787663,0.00007092474,0.00010372117,0.000012209649,0.000007975884,0.000024693118,0.000021776517,0.97550565,0.0026914733,0.020246796,0.0012676863,0.000009291287],"about_ca_topic_score_codex":0.0017497669,"about_ca_topic_score_gemma":0.0019212837,"teacher_disagreement_score":0.0018392518,"about_ca_system_score_codex":0.00060842576,"about_ca_system_score_gemma":0.00075703824,"threshold_uncertainty_score":0.0063536167},"labels":[],"label_agreement":null},{"id":"W2545079506","doi":"10.1109/iat.2007.74","title":"Reinforcement Learning with Inertial Exploration","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Reinforcement learning; Abstraction; Computer science; Context (archaeology); Action selection; Inertial frame of reference; Set (abstract data type); Artificial intelligence; Selection (genetic algorithm); Scale (ratio); Control (management); Machine learning; Action (physics)","score_opus":0.018947262859378568,"score_gpt":0.24544685046786655,"score_spread":0.22649958760848798,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2545079506","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0131728975,0.000283568,0.9835195,0.00023807694,0.000048356,0.00004180212,0.000015226268,0.00025011762,0.0024304932],"genre_scores_gemma":[0.8769835,0.00025947782,0.11865192,0.00022219558,0.00012467419,0.0002616676,0.000048962436,0.000040881645,0.0034066446],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99921,0.00037821828,0.000033730004,0.00013175765,0.00016677457,0.00007955796],"domain_scores_gemma":[0.99785346,0.0013897589,0.00020219879,0.00019702007,0.00022759243,0.00012995138],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015628977,0.0008268055,0.00091767457,0.00031590607,0.00033039012,0.00071811496,0.0011721238,0.0009289436,0.0022224032],"category_scores_gemma":[0.005913945,0.00030372592,0.00039304205,0.00038457758,0.00155906,0.0013197875,0.0013635708,0.0012679357,0.0003573755],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015745792,0.0001270444,0.0009197583,0.00010304258,0.00006201697,0.00010192935,0.00009942706,0.8480144,0.0010798507,0.076689124,0.001593364,0.07105258],"study_design_scores_gemma":[0.000030457873,0.00005211347,0.00006445036,0.0000043611026,0.0000043962586,0.0000099516465,0.000004015334,0.9818357,0.00016684264,0.017382892,0.000440502,0.000004324314],"about_ca_topic_score_codex":0.0023609698,"about_ca_topic_score_gemma":0.0012721397,"teacher_disagreement_score":0.0023609698,"about_ca_system_score_codex":0.0006922652,"about_ca_system_score_gemma":0.0009338377,"threshold_uncertainty_score":0.008265495},"labels":[],"label_agreement":null},{"id":"W2546975091","doi":"","title":"Learning Locomotion Skills Using DeepRL: Does the Choice of Action Space Matter?","year":2016,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":68,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Robustness (evolution); Reinforcement learning; Computer science; Artificial intelligence; Torque; Machine learning; Physics","score_opus":0.0546902574138028,"score_gpt":0.20722812833932142,"score_spread":0.15253787092551863,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2546975091","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6608105,0.0015922582,0.3297504,0.0019171606,0.00009655191,0.00010083729,0.00016726389,0.0011137481,0.0044513443],"genre_scores_gemma":[0.980911,0.00019602684,0.017980902,0.00012580822,0.000009532539,0.000026402744,0.00009147699,0.000038014405,0.0006207738],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99935836,0.00025764803,0.000039544288,0.00014435209,0.00008830818,0.000111784764],"domain_scores_gemma":[0.99647266,0.002413504,0.00036565177,0.00030210218,0.00024393875,0.00020212513],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028963292,0.00057172775,0.00066777895,0.00022377238,0.00018788913,0.0009252681,0.00081890175,0.000900765,0.0014203165],"category_scores_gemma":[0.012219983,0.00033741456,0.00028179,0.00019583278,0.0008901008,0.0020966101,0.00080072274,0.0016030784,0.0003769353],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00067272875,0.00050752104,0.016113484,0.00031106468,0.00018511088,0.00009551365,0.00011707365,0.67875594,0.012383523,0.0048199003,0.0015197691,0.28451845],"study_design_scores_gemma":[0.00003318812,0.00014234114,0.0012927996,0.000026123118,0.000015843369,0.000016979744,0.000024525469,0.9909552,0.0024086456,0.0048939623,0.00018244555,0.000007975934],"about_ca_topic_score_codex":0.0032724396,"about_ca_topic_score_gemma":0.0036991793,"teacher_disagreement_score":0.0032724396,"about_ca_system_score_codex":0.0006137429,"about_ca_system_score_gemma":0.000838961,"threshold_uncertainty_score":0.01531738},"labels":[],"label_agreement":null},{"id":"W2552186722","doi":"10.1109/fuzz-ieee.2016.7737798","title":"Reinforcement learning in the guarding a territory game","year":2016,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Guard (computer science); Reinforcement learning; Computer science; Game theory; A priori and a posteriori; Set (abstract data type); Artificial intelligence; Repeated game; Non-cooperative game; Mathematical economics; Mathematics","score_opus":0.016922623687089227,"score_gpt":0.24108977273666846,"score_spread":0.22416714904957924,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2552186722","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.30415136,0.00031905348,0.6860883,0.0007247552,0.000058106947,0.00015926521,0.00003965835,0.00028897432,0.008170476],"genre_scores_gemma":[0.9719283,0.00007515967,0.026128842,0.00009338726,0.000017779164,0.00008014237,0.00001776127,0.000016363058,0.0016422253],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99890876,0.00062524976,0.00003081117,0.00014870499,0.00013993034,0.00014655557],"domain_scores_gemma":[0.9943551,0.004491213,0.00040110297,0.000119100216,0.00029131828,0.00034214134],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002267978,0.0007948197,0.0009746797,0.00031941364,0.0004343079,0.0006228805,0.0012506942,0.001101215,0.0019266851],"category_scores_gemma":[0.008298204,0.00028821544,0.00037062835,0.00019897679,0.001940819,0.001319465,0.00093497644,0.001418497,0.0001741824],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022595242,0.00014343984,0.0011772099,0.000053280415,0.000033149896,0.000118531636,0.000101519974,0.9697356,0.0009144111,0.015166372,0.00030011006,0.012030425],"study_design_scores_gemma":[0.000029706158,0.00007069938,0.00011048119,0.000004017777,0.000004304764,0.000011223219,0.000013566802,0.99420905,0.00021420258,0.005201763,0.00012604275,0.0000050413923],"about_ca_topic_score_codex":0.006681316,"about_ca_topic_score_gemma":0.0040227934,"teacher_disagreement_score":0.006681316,"about_ca_system_score_codex":0.0011196064,"about_ca_system_score_gemma":0.0012465863,"threshold_uncertainty_score":0.013284862},"labels":[],"label_agreement":null},{"id":"W2554885577","doi":"10.1109/ijcnn.2016.7727651","title":"Context-switching and adaptation: Brain-inspired mechanisms for handling environmental changes","year":2016,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Reinforcement learning; Computer science; Abstraction; Adaptation (eye); Context (archaeology); Task (project management); Artificial intelligence; Process (computing); Transfer of learning; Grid; State (computer science); Machine learning; Neuroscience; Engineering; Algorithm","score_opus":0.02596948343333376,"score_gpt":0.22451772963762429,"score_spread":0.1985482462042905,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2554885577","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19049935,0.0009272812,0.7965406,0.0006126252,0.00017765739,0.00009506286,0.00006448687,0.0016185388,0.009464378],"genre_scores_gemma":[0.9500878,0.00030221607,0.048282735,0.0001376187,0.00003135699,0.000083008585,0.000033979242,0.000054707954,0.0009866401],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998111,0.00005684686,0.000011050567,0.00005006065,0.00004173185,0.000029205132],"domain_scores_gemma":[0.9994011,0.0002862858,0.00009626665,0.00011388921,0.00004495674,0.000057561345],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00039147245,0.00045804822,0.00032771612,0.00024774752,0.00026767212,0.0005770383,0.0010671898,0.00061724393,0.0017370965],"category_scores_gemma":[0.0023831653,0.00023458026,0.0004874727,0.00019979045,0.0010096773,0.0010986903,0.0009231731,0.00092707406,0.00019657999],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026924428,0.00023308897,0.0030892058,0.00021606787,0.00023009792,0.00030796923,0.00029994707,0.67215794,0.0673532,0.06266866,0.0021363911,0.19103815],"study_design_scores_gemma":[0.000058500365,0.00014815878,0.0010348365,0.00001665201,0.00004856404,0.00011847019,0.000028123603,0.92651343,0.011554035,0.057802375,0.0026416609,0.00003524309],"about_ca_topic_score_codex":0.0012284858,"about_ca_topic_score_gemma":0.0010490763,"teacher_disagreement_score":0.0017370965,"about_ca_system_score_codex":0.00040271934,"about_ca_system_score_gemma":0.00041129783,"threshold_uncertainty_score":0.005811155},"labels":[],"label_agreement":null},{"id":"W2559277598","doi":"10.1523/jneurosci.0763-16.2016","title":"Memory Transformation Enhances Reinforcement Learning in Dynamic Environments","year":2016,"lang":"en","type":"article","venue":"Journal of Neuroscience","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"The Scarborough Hospital; Hospital for Sick Children; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Schematic; Computer science; Episodic memory; Memory consolidation; Salience (neuroscience); Adaptive memory; Transformation (genetics); Artificial intelligence; Cognition; Psychology; Neuroscience","score_opus":0.014104161412884272,"score_gpt":0.2480868169701692,"score_spread":0.23398265555728492,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2559277598","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8664551,0.00015405317,0.12568533,0.0002937847,0.000045025226,0.000029123054,0.00004404283,0.0007894299,0.0065041594],"genre_scores_gemma":[0.99065065,0.00004088858,0.008637719,0.000033461743,0.0000037855027,0.000009571894,0.000026376023,0.000018884764,0.00057850045],"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9999045,0.00001916973,0.0000064323262,0.000029881548,0.000022396836,0.000017534838],"domain_scores_gemma":[0.99962497,0.00012635211,0.00007897358,0.000077651166,0.00004461003,0.00004742183],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00021237928,0.00023394013,0.00020213773,0.00009724801,0.000109883804,0.00046390895,0.000456639,0.00032013922,0.0018734639],"category_scores_gemma":[0.0013382756,0.000120832476,0.00023688967,0.00009032396,0.00039251504,0.00091665564,0.000631207,0.0005331417,0.00022104073],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005637924,0.0010211181,0.0090677915,0.00024992143,0.00012590423,0.00054813846,0.00036225474,0.3389385,0.3867795,0.034929235,0.0010430289,0.22637084],"study_design_scores_gemma":[0.000055774453,0.0009504893,0.0061660702,0.000015834637,0.00005098713,0.0002516131,0.00006659648,0.85435295,0.097292356,0.037615445,0.0031500373,0.000031853484],"about_ca_topic_score_codex":0.0003821143,"about_ca_topic_score_gemma":0.0004286013,"teacher_disagreement_score":0.0018734639,"about_ca_system_score_codex":0.00026261364,"about_ca_system_score_gemma":0.00027153204,"threshold_uncertainty_score":0.006267369},"labels":[],"label_agreement":null},{"id":"W2561828331","doi":"10.1609/aaai.v30i1.10304","title":"Compressed Conditional Mean Embeddings for Model-Based Reinforcement Learning","year":2016,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Reinforcement learning; Markov decision process; Class (philosophy); Context (archaeology); Kernel (algebra); Bottleneck; Mathematical optimization; Artificial intelligence; Machine learning; Markov process; Mathematics","score_opus":0.07598492911930667,"score_gpt":0.3017610321490101,"score_spread":0.22577610302970347,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2561828331","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0063626594,0.000096706644,0.9920339,0.00015493373,0.000023801042,0.000017869612,0.000043528536,0.00041398293,0.0008525567],"genre_scores_gemma":[0.7011598,0.00022178816,0.2948527,0.0002210396,0.000075067364,0.00024621928,0.0003190705,0.00023810694,0.0026662608],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994661,0.00020489313,0.00002444471,0.00010474752,0.00014437934,0.0000554472],"domain_scores_gemma":[0.9976484,0.0016075511,0.00021676146,0.00024442977,0.00018086848,0.0001019139],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010468604,0.0008985607,0.0011597693,0.0004722524,0.00031600246,0.0007205551,0.001437333,0.0011112272,0.0033293164],"category_scores_gemma":[0.006229788,0.00054804207,0.00067127944,0.000472966,0.0011062667,0.0017639407,0.0017549912,0.0026933993,0.00045647012],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000048907103,0.000040749663,0.0002082527,0.000043676333,0.000019264608,0.000028905593,0.00003201474,0.9488815,0.0006134252,0.0281145,0.0007699412,0.021198813],"study_design_scores_gemma":[0.0000031625666,0.00000851326,0.000010548114,0.000002032755,0.000001303836,0.00000364653,0.0000011163368,0.99164164,0.00014393197,0.008048518,0.00013353796,0.0000020336608],"about_ca_topic_score_codex":0.003391833,"about_ca_topic_score_gemma":0.0028271475,"teacher_disagreement_score":0.003391833,"about_ca_system_score_codex":0.0010419536,"about_ca_system_score_gemma":0.0012791065,"threshold_uncertainty_score":0.011137664},"labels":[],"label_agreement":null},{"id":"W2567143531","doi":"10.6084/m9.figshare.3085861.v1","title":"Reinforcement Learning in a Nutshell","year":2016,"lang":"en","type":"article","venue":"Figshare","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Reinforcement; Artificial intelligence; Reinforcement learning; Computer science; Psychology; Social psychology","score_opus":0.029458926697487377,"score_gpt":0.24673300033253034,"score_spread":0.21727407363504297,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2567143531","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0012797905,0.00052769564,0.987225,0.0009801126,0.00023340827,0.00007279075,0.00007509393,0.0026037558,0.0070023155],"genre_scores_gemma":[0.11501645,0.0015261119,0.8527375,0.0012984631,0.00035030028,0.00049527624,0.00038343386,0.0012932775,0.02689921],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9993594,0.00016108787,0.00005883817,0.00014848617,0.00023356768,0.00003856505],"domain_scores_gemma":[0.99862194,0.0004903358,0.000070341324,0.00041268938,0.00030092045,0.00010373878],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015436905,0.0010252359,0.0011335842,0.00041596044,0.0003558575,0.0017882992,0.002035311,0.0015033359,0.01595013],"category_scores_gemma":[0.0053529916,0.00079767726,0.00075663585,0.0003520531,0.0011935323,0.003331046,0.0027071482,0.00415306,0.009433576],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027558283,0.00018887428,0.0006985704,0.0004790547,0.00024948196,0.00018072216,0.00020318569,0.166662,0.007921016,0.29229763,0.029231112,0.5016127],"study_design_scores_gemma":[0.000089618836,0.00017039916,0.00021049274,0.000263256,0.00006836632,0.00022404979,0.000038111513,0.46702605,0.0053878794,0.45874807,0.067708515,0.000065170934],"about_ca_topic_score_codex":0.0009833111,"about_ca_topic_score_gemma":0.0012708767,"teacher_disagreement_score":0.01595013,"about_ca_system_score_codex":0.0005356197,"about_ca_system_score_gemma":0.0010366465,"threshold_uncertainty_score":0.053358555},"labels":[],"label_agreement":null},{"id":"W2568999269","doi":"10.1609/aaai.v30i1.10311","title":"Incremental Stochastic Factorization for Online Reinforcement Learning","year":2016,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Markov decision process; Reinforcement learning; Computer science; Non-negative matrix factorization; Probabilistic logic; Multiplicative function; Factorization; Divergence (linguistics); Bellman equation; Markov chain; Artificial intelligence; Markov process; Algorithm; Mathematical optimization; Theoretical computer science; Machine learning; Matrix decomposition; Mathematics; Eigenvalues and eigenvectors","score_opus":0.0805954660865807,"score_gpt":0.3016254592844599,"score_spread":0.2210299931978792,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2568999269","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0025191996,0.00029696326,0.9951029,0.00014586344,0.000048013306,0.00004198716,0.000041560535,0.00026025175,0.0015432932],"genre_scores_gemma":[0.6152559,0.0009133901,0.3754227,0.0003531865,0.00021975761,0.00069744914,0.00038439702,0.00016267455,0.0065905377],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991328,0.00033014626,0.000043872646,0.00018608116,0.00022274694,0.000084446976],"domain_scores_gemma":[0.99751747,0.001804319,0.00017673247,0.00015804652,0.00025490916,0.00008851379],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021666938,0.0013599123,0.0017024145,0.00064452505,0.00042154797,0.0008257763,0.0014480823,0.001172001,0.0061235065],"category_scores_gemma":[0.0074801273,0.000581144,0.0008201411,0.00071772636,0.0011043088,0.001296226,0.0011227395,0.0022679425,0.00081099564],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008846946,0.000083039806,0.000448644,0.00016611865,0.00006207875,0.0000816758,0.00006157673,0.849345,0.00086179294,0.08141572,0.0027797483,0.06460611],"study_design_scores_gemma":[0.000010335289,0.000017486558,0.000029946676,0.000006324949,0.0000048498027,0.0000068724275,0.0000020502735,0.9773119,0.000110966146,0.021898752,0.0005959977,0.0000044912795],"about_ca_topic_score_codex":0.0061714747,"about_ca_topic_score_gemma":0.0070100524,"teacher_disagreement_score":0.0061714747,"about_ca_system_score_codex":0.0015454846,"about_ca_system_score_gemma":0.0017769826,"threshold_uncertainty_score":0.020485163},"labels":[],"label_agreement":null},{"id":"W2570416447","doi":"10.1007/978-3-642-33093-3_30","title":"Multi-timescale Nexting in a Reinforcement Learning Robot","year":2012,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Reinforcement learning; Robot; Artificial intelligence; Human–computer interaction","score_opus":0.02659981860661804,"score_gpt":0.2579270346567551,"score_spread":0.23132721605013706,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2570416447","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1303961,0.00050590024,0.8500818,0.00066810177,0.0002906252,0.000087885535,0.000051229315,0.0016465022,0.01627183],"genre_scores_gemma":[0.8836704,0.00014612645,0.10618704,0.00009702831,0.0000325916,0.00006122394,0.00002947131,0.00006744545,0.009708718],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998673,0.00002379708,0.000007498102,0.00004705486,0.00003105708,0.000023221979],"domain_scores_gemma":[0.9997255,0.00011093931,0.000025650179,0.00003677507,0.000038491085,0.00006266558],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00037951677,0.00033495395,0.0005418068,0.00018429382,0.00066654856,0.00045374635,0.0011220168,0.00097426056,0.003905439],"category_scores_gemma":[0.00068538456,0.00030242855,0.0003398766,0.00017212458,0.0005819236,0.00081665406,0.001054805,0.0009557451,0.00042490524],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005910799,0.00030492558,0.0013864537,0.000151222,0.000054947486,0.0006740193,0.0002099686,0.76565325,0.05705065,0.034015,0.002190212,0.13771825],"study_design_scores_gemma":[0.000018611541,0.00010604461,0.00019656985,0.00000604558,0.0000092846685,0.000050988354,0.0000118860125,0.9906227,0.0024114626,0.00557065,0.0009841064,0.000011721149],"about_ca_topic_score_codex":0.0024359955,"about_ca_topic_score_gemma":0.0024531481,"teacher_disagreement_score":0.003905439,"about_ca_system_score_codex":0.00042143546,"about_ca_system_score_gemma":0.0005112343,"threshold_uncertainty_score":0.01306504},"labels":[],"label_agreement":null},{"id":"W2574782009","doi":"","title":"Learning multi-step predictive state representations","year":2016,"lang":"en","type":"article","venue":"International Joint Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Exploit; Observable; Dynamical systems theory; Variety (cybernetics); Representation (politics); Artificial intelligence; Class (philosophy); Machine learning; State (computer science); Theoretical computer science; Algorithm","score_opus":0.12193993464212904,"score_gpt":0.3447195660178371,"score_spread":0.22277963137570805,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2574782009","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023757825,0.00016295852,0.9742882,0.00012006196,0.000022448558,0.000026930074,0.00008517526,0.0007734102,0.000762929],"genre_scores_gemma":[0.84657204,0.00017519708,0.15089175,0.00012666151,0.000045056644,0.00017414082,0.0003872814,0.00007705214,0.0015508406],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993864,0.0001821788,0.000037515532,0.00017390902,0.00015798245,0.000061957035],"domain_scores_gemma":[0.99700385,0.002030876,0.0002711647,0.00037082093,0.00025596542,0.000067270346],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013715083,0.0006202491,0.0010931159,0.00062161055,0.0002788276,0.0009595517,0.0015903072,0.0011004647,0.0016751295],"category_scores_gemma":[0.0067141424,0.00053773774,0.00063298515,0.00071937323,0.00073032867,0.0017057916,0.0011951779,0.0015636978,0.00041958233],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007008969,0.00004094753,0.0006183952,0.000042959284,0.00003166302,0.000043305965,0.000050156137,0.92359906,0.000848186,0.0097566275,0.00052004313,0.064378604],"study_design_scores_gemma":[0.0000020186326,0.000006433286,0.000029993165,0.0000017520028,0.0000015326953,0.000003255527,0.0000015351937,0.9977883,0.00013857501,0.0019826712,0.000042268835,0.0000017307659],"about_ca_topic_score_codex":0.0027493264,"about_ca_topic_score_gemma":0.0028809234,"teacher_disagreement_score":0.0027493264,"about_ca_system_score_codex":0.00060509884,"about_ca_system_score_gemma":0.0007250327,"threshold_uncertainty_score":0.007253289},"labels":[],"label_agreement":null},{"id":"W2579061194","doi":"","title":"Pairwise Relative Offset Features for Atari 2600 Games.","year":2015,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Pairwise comparison; Computer science; Offset (computer science); Artificial intelligence; Feature (linguistics); Reinforcement learning; Computer vision; Set (abstract data type); Pattern recognition (psychology)","score_opus":0.04626850614696447,"score_gpt":0.281039109827256,"score_spread":0.2347706036802915,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2579061194","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07116449,0.00086718134,0.89601433,0.00041061814,0.00034516843,0.00052778295,0.00171855,0.007450681,0.021501252],"genre_scores_gemma":[0.66542745,0.00020644146,0.31941983,0.00028165232,0.0000612613,0.00069784984,0.0028555945,0.00037962705,0.0106703695],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995171,0.00010302384,0.000023084564,0.00010979387,0.00019695633,0.000050022838],"domain_scores_gemma":[0.9994167,0.00019583269,0.000078611956,0.00011133115,0.00011741969,0.00008006739],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005680174,0.0012427008,0.00071764976,0.00047296146,0.0003142956,0.0006597894,0.0017190409,0.0008078172,0.00642349],"category_scores_gemma":[0.0029952487,0.00028902746,0.00051082484,0.00031603864,0.00042839174,0.0011042962,0.0011596947,0.001615992,0.0019803727],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008983942,0.00088645157,0.0035042139,0.0003692023,0.00012488994,0.00022627892,0.0000985938,0.26894292,0.025007479,0.02626158,0.028514383,0.6451657],"study_design_scores_gemma":[0.000065617314,0.00038916423,0.0015452242,0.000027448954,0.00001577756,0.00010330478,0.000017524675,0.9670036,0.006685052,0.014141612,0.00996835,0.000037438178],"about_ca_topic_score_codex":0.0043309713,"about_ca_topic_score_gemma":0.0078716595,"teacher_disagreement_score":0.00642349,"about_ca_system_score_codex":0.00078619877,"about_ca_system_score_gemma":0.00085696595,"threshold_uncertainty_score":0.021488786},"labels":[],"label_agreement":null},{"id":"W2593237273","doi":"10.1609/aaai.v32i1.11631","title":"Multi-Step Reinforcement Learning: A Unifying Algorithm","year":2018,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":107,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates; DeepMind","keywords":"Reinforcement learning; Computer science; Backup; Algorithm; TRACE (psycholinguistics); Sampling (signal processing); Focus (optics); Monte Carlo method; Importance sampling; Artificial intelligence; Machine learning; Mathematics","score_opus":0.09247718432083181,"score_gpt":0.31363613619094266,"score_spread":0.22115895187011086,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2593237273","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0043565007,0.00012386667,0.99379635,0.00017328857,0.00002700765,0.00005120847,0.000009462159,0.00049820414,0.0009641481],"genre_scores_gemma":[0.26403397,0.00020196322,0.73253405,0.00030177055,0.0000790517,0.00026469538,0.00006869081,0.00021737716,0.0022983516],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99790597,0.000637068,0.00014626875,0.00059276674,0.00054718833,0.00017068745],"domain_scores_gemma":[0.99623775,0.0022688603,0.0002181543,0.00069558504,0.0003746171,0.00020501325],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051256567,0.0010897092,0.0016828169,0.00083400647,0.00058709964,0.0013436697,0.0033492136,0.002242376,0.0021321499],"category_scores_gemma":[0.010697602,0.00059134385,0.0007992157,0.0006536894,0.001699375,0.0030623088,0.0029024207,0.003134806,0.00063576724],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022776281,0.00022486318,0.0013231089,0.000122043624,0.000075025084,0.000070562135,0.00022952475,0.6289505,0.0030763058,0.100575306,0.0018595569,0.2632655],"study_design_scores_gemma":[0.000024348688,0.00004901222,0.000036078978,0.000009986517,0.000006142607,0.00001603444,0.000004921979,0.98259616,0.0006418874,0.015958874,0.0006500112,0.0000065829568],"about_ca_topic_score_codex":0.0021413197,"about_ca_topic_score_gemma":0.0020102465,"teacher_disagreement_score":0.0051256567,"about_ca_system_score_codex":0.0012756105,"about_ca_system_score_gemma":0.00221014,"threshold_uncertainty_score":0.027107358},"labels":[],"label_agreement":null},{"id":"W2594903727","doi":"10.48550/arxiv.1702.08360","title":"Neural Map: Structured Memory for Deep Reinforcement Learning","year":2017,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":102,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Observability; Artificial intelligence; Set (abstract data type); Artificial neural network; Simple (philosophy); Architecture; Memory map; Deep learning; Shared memory; Parallel computing; Programming language","score_opus":0.0666945800872272,"score_gpt":0.2102999486985228,"score_spread":0.1436053686112956,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2594903727","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018884828,0.00032487395,0.97466666,0.00028861756,0.00009025752,0.00004001774,0.0001967086,0.0026477713,0.0028602884],"genre_scores_gemma":[0.79493713,0.0002878614,0.19854218,0.00020544273,0.000049622875,0.00024090675,0.00036498447,0.00017506444,0.005196866],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99985075,0.00003380147,0.0000081632,0.000043367767,0.000041664513,0.000022189563],"domain_scores_gemma":[0.9996301,0.00016525337,0.000037859296,0.00006660152,0.000067576184,0.000032573415],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004211208,0.0006833353,0.00055493857,0.00025099478,0.0002019836,0.0006200006,0.0015671573,0.00084977306,0.0043210797],"category_scores_gemma":[0.0020687033,0.0003187584,0.00037388867,0.00023990887,0.00053615577,0.0010388345,0.0010928962,0.0014140098,0.000771816],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020854716,0.00010574281,0.0008845913,0.00014953311,0.000078894285,0.00014543008,0.000076157594,0.7823646,0.006777955,0.028097674,0.005828615,0.17528225],"study_design_scores_gemma":[0.000008890415,0.000026507298,0.0000491022,0.0000043490045,0.000003844474,0.0000105605695,0.000003590888,0.98769605,0.001263774,0.010242271,0.0006877733,0.0000033499432],"about_ca_topic_score_codex":0.0025704168,"about_ca_topic_score_gemma":0.0032728938,"teacher_disagreement_score":0.0043210797,"about_ca_system_score_codex":0.0005590847,"about_ca_system_score_gemma":0.0006332793,"threshold_uncertainty_score":0.014455438},"labels":[],"label_agreement":null},{"id":"W2605760243","doi":"","title":"Teaching with RoboCup","year":2004,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Computer science; Artificial intelligence","score_opus":0.009273550210222827,"score_gpt":0.22538821041693533,"score_spread":0.2161146602067125,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2605760243","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.054993976,0.001554528,0.017865844,0.005675235,0.0038474551,0.001127454,0.008755181,0.015035051,0.89114535],"genre_scores_gemma":[0.17949986,0.0016534079,0.020076785,0.0016416186,0.00058648706,0.0009378291,0.01264171,0.002124694,0.78083754],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99925524,0.00012798265,0.000041574385,0.00012384613,0.0003160465,0.00013532626],"domain_scores_gemma":[0.9976273,0.00032834482,0.00008923827,0.0003404188,0.00061975955,0.0009950366],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009821927,0.00085541635,0.0005999085,0.0009172418,0.0010474869,0.0032349727,0.0017461476,0.00073040905,0.2934421],"category_scores_gemma":[0.004701651,0.00020918706,0.0005648502,0.000727987,0.00045895492,0.0018238305,0.0019417182,0.0008308343,0.11531516],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004148885,0.0009256553,0.0017024871,0.0005871721,0.000026535028,0.00018095826,0.0006423759,0.0010120453,0.0025151437,0.0038666644,0.51219565,0.47593048],"study_design_scores_gemma":[0.00013834113,0.00069060293,0.007927481,0.00035705316,0.000034249635,0.0002647436,0.0011711637,0.0017055527,0.006157191,0.0071741524,0.9743333,0.00004623205],"about_ca_topic_score_codex":0.002815059,"about_ca_topic_score_gemma":0.009712828,"teacher_disagreement_score":0.2934421,"about_ca_system_score_codex":0.0010780114,"about_ca_system_score_gemma":0.0015714661,"threshold_uncertainty_score":0.9816616},"labels":[],"label_agreement":null},{"id":"W2607014226","doi":"10.1109/tnnls.2017.2690910","title":"Learning to Predict Consequences as a Method of Knowledge Transfer in Reinforcement Learning","year":2017,"lang":"en","type":"article","venue":"IEEE Transactions on Neural Networks and Learning Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Reinforcement learning; Affordance; Robot; Human–computer interaction; Artificial intelligence; Robot learning; Semantics (computer science); Knowledge transfer; Transfer of learning; Task (project management); Knowledge management; Mobile robot","score_opus":0.025031320524130575,"score_gpt":0.2901557942169168,"score_spread":0.2651244736927862,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2607014226","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010904409,0.00012509129,0.98621386,0.00024346019,0.000035486868,0.00009513648,0.000020609132,0.00030266735,0.002059265],"genre_scores_gemma":[0.7470756,0.00024176038,0.2491998,0.00023062095,0.000068896385,0.0005154939,0.00006184879,0.00007256464,0.002533395],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982571,0.00084732636,0.00010092037,0.00030223708,0.0003905595,0.0001019054],"domain_scores_gemma":[0.9936566,0.0044726375,0.0004280683,0.0007528794,0.00050822133,0.0001816732],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038693028,0.0011109636,0.0010948756,0.00063727685,0.00045306692,0.001040366,0.0023716819,0.0015578534,0.0025934994],"category_scores_gemma":[0.013700375,0.00047614807,0.0007245685,0.0006320734,0.0024450526,0.002548899,0.0020244122,0.0024783246,0.00039198573],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016946161,0.00024022862,0.0015854766,0.00013705579,0.00012281467,0.0001908527,0.00028759034,0.79186684,0.0025926454,0.06750384,0.0012466345,0.13405655],"study_design_scores_gemma":[0.00003562491,0.00007121575,0.000119127864,0.000011551958,0.0000152135635,0.000026166275,0.000010917186,0.96013516,0.0009151027,0.038120985,0.0005260739,0.000012867123],"about_ca_topic_score_codex":0.002550864,"about_ca_topic_score_gemma":0.001980878,"teacher_disagreement_score":0.0038693028,"about_ca_system_score_codex":0.001157633,"about_ca_system_score_gemma":0.0011021228,"threshold_uncertainty_score":0.020463109},"labels":[],"label_agreement":null},{"id":"W2620853660","doi":"10.5555/3091125.3091207","title":"Forward Actor-Critic for Nonlinear Function Approximation in Reinforcement Learning","year":2017,"lang":"en","type":"article","venue":"Adaptive Agents and Multi-Agents Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Bellman equation; Function approximation; Nonlinear system; Function (biology); Class (philosophy); Artificial intelligence; Temporal difference learning; Mathematical optimization; Machine learning; Artificial neural network; Mathematics","score_opus":0.07223075096191683,"score_gpt":0.309418597286914,"score_spread":0.23718784632499718,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2620853660","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0027293153,0.00037397165,0.99448746,0.00013420555,0.000050423303,0.000030148007,0.000015477037,0.00026100915,0.0019180207],"genre_scores_gemma":[0.65965873,0.0008231437,0.32714632,0.00029092986,0.000105466235,0.00038019838,0.00010920644,0.00019535665,0.011290703],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994042,0.00023461516,0.000031898213,0.000115263385,0.00016559586,0.000048427395],"domain_scores_gemma":[0.9981445,0.0012830356,0.00011430953,0.00013936851,0.00026129058,0.000057427158],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020957452,0.0012729142,0.0011402239,0.0004254496,0.0003907899,0.0008706378,0.0013603838,0.0014612275,0.0030324159],"category_scores_gemma":[0.005353895,0.00061772886,0.0006460789,0.00045605176,0.0013481419,0.0008851603,0.0010133693,0.0026996315,0.00068500516],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005019353,0.000031403088,0.00032816493,0.00009744124,0.000041205578,0.00006907418,0.000054380373,0.932126,0.0011050161,0.029808369,0.0010651746,0.03522356],"study_design_scores_gemma":[0.000005012286,0.000008953005,0.000018249872,0.0000043135774,0.0000030375027,0.000005740419,0.00000127468,0.9954477,0.00020348458,0.003933962,0.00036545927,0.0000027739932],"about_ca_topic_score_codex":0.0052340776,"about_ca_topic_score_gemma":0.005072001,"teacher_disagreement_score":0.0052340776,"about_ca_system_score_codex":0.0011307881,"about_ca_system_score_gemma":0.0012556661,"threshold_uncertainty_score":0.011083484},"labels":[],"label_agreement":null},{"id":"W2626637010","doi":"10.4230/lipics.cp.2023.25","title":"Learning a Generic Value-Selection Heuristic Inside a Constraint Programming Solver","year":2017,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":523,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Decomposition; Value (mathematics); Computer science; Artificial intelligence; Machine learning; Chemistry","score_opus":0.07875522306874225,"score_gpt":0.20925438421090964,"score_spread":0.1304991611421674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2626637010","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009355835,0.00009579103,0.98665476,0.0002610932,0.00003079299,0.00008565192,0.000047674654,0.00041086346,0.0030575178],"genre_scores_gemma":[0.38691738,0.00018843646,0.6085796,0.000396389,0.00006947942,0.000410825,0.00021271664,0.00023866886,0.002986515],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986004,0.00056957843,0.00007332216,0.00032936197,0.00023603607,0.00019127474],"domain_scores_gemma":[0.9962202,0.0027228505,0.00030495203,0.00029131575,0.00029674688,0.0001639758],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024890974,0.0012315806,0.0012386038,0.0006683621,0.00042007718,0.0014795032,0.0019513884,0.002238475,0.0051420988],"category_scores_gemma":[0.00906559,0.00060814666,0.00089054997,0.00090633123,0.0015319451,0.0016581732,0.0018833104,0.0025511587,0.0008429335],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001120283,0.00013325212,0.00085514726,0.00018555053,0.00005122298,0.00013356523,0.00009790236,0.8495513,0.0021041927,0.06050227,0.003111804,0.08316179],"study_design_scores_gemma":[0.000020151878,0.000022647513,0.000043789612,0.0000141740875,0.000006330248,0.000014917902,0.000008780971,0.9827047,0.00060539605,0.015963273,0.00059050927,0.0000054127513],"about_ca_topic_score_codex":0.0018777453,"about_ca_topic_score_gemma":0.0025118757,"teacher_disagreement_score":0.0051420988,"about_ca_system_score_codex":0.0012081687,"about_ca_system_score_gemma":0.0023151767,"threshold_uncertainty_score":0.01720202},"labels":[],"label_agreement":null},{"id":"W2724228633","doi":"10.1145/3071178.3071303","title":"Multi-task learning in Atari video games with emergent tangled program graphs","year":2017,"lang":"en","type":"article","venue":"Proceedings of the Genetic and Evolutionary Computation Conference","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Observability; Task (project management); Video game; Artificial intelligence; Variety (cybernetics); State (computer science); Matching (statistics); Human–computer interaction; Machine learning; Multimedia; Programming language","score_opus":0.021923593551869416,"score_gpt":0.25637648094619514,"score_spread":0.23445288739432574,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2724228633","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.673693,0.0003122867,0.3190511,0.0006524448,0.000040745734,0.00014770664,0.00010409273,0.0004844424,0.005514212],"genre_scores_gemma":[0.95996463,0.000050219907,0.03750973,0.00007492577,0.000008641674,0.000098838194,0.00009998056,0.000048956455,0.002143975],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997274,0.000102458056,0.000010936877,0.00006940166,0.000036070378,0.0000536602],"domain_scores_gemma":[0.9983638,0.0012006762,0.00011254488,0.00005769616,0.00009752933,0.00016780752],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00086323475,0.00063733547,0.0005902492,0.00044741118,0.00042433917,0.00060910726,0.000986762,0.00080249563,0.001492762],"category_scores_gemma":[0.0045538954,0.00040616654,0.0004474894,0.00022528032,0.0008040932,0.0010373534,0.0010926115,0.0011319498,0.00011325036],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007863471,0.00011024694,0.0011617162,0.00003360403,0.000021329259,0.00011920139,0.000116715375,0.97623676,0.0008620608,0.006080222,0.00050806836,0.014671401],"study_design_scores_gemma":[0.000008932842,0.000017393599,0.00013404319,0.000002272369,0.0000017180704,0.0000051891875,0.000014412801,0.99538016,0.00011926188,0.004224124,0.000090637026,0.0000019226031],"about_ca_topic_score_codex":0.010382884,"about_ca_topic_score_gemma":0.013720197,"teacher_disagreement_score":0.010382884,"about_ca_system_score_codex":0.0012800163,"about_ca_system_score_gemma":0.00069620943,"threshold_uncertainty_score":0.020644903},"labels":[],"label_agreement":null},{"id":"W2733958039","doi":"10.1609/aiide.v11i1.12801","title":"Maximizing Flow as a Metacontrol in Angband","year":2015,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"University of Alberta","keywords":"Computer science; State (computer science); Flow (mathematics); Artificial intelligence; Operations research; Psychology; Engineering; Mathematics","score_opus":0.05941970819663916,"score_gpt":0.28596994665065184,"score_spread":0.22655023845401268,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2733958039","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.69805723,0.00022917415,0.26073763,0.0007416319,0.00007609375,0.00018989207,0.000083801766,0.00086489343,0.039019603],"genre_scores_gemma":[0.976513,0.000039741724,0.0210432,0.0000803978,0.000007052751,0.000075598364,0.000022943676,0.000032681957,0.0021853677],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99942833,0.00026538805,0.000022401846,0.00009311999,0.000083664454,0.00010707839],"domain_scores_gemma":[0.9985618,0.00066945027,0.000256111,0.00011778932,0.00009827663,0.00029649874],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011485924,0.00061010366,0.00030976767,0.00037228546,0.000351241,0.0012440734,0.00047266472,0.00056480954,0.0032876784],"category_scores_gemma":[0.004460566,0.00017699912,0.00019618377,0.00014312728,0.0008241931,0.0013981182,0.0012322582,0.0006469526,0.00029647365],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028298378,0.002480176,0.03023759,0.0005254681,0.00020943128,0.00093395024,0.004152432,0.31651938,0.112876765,0.26352957,0.005252176,0.26045328],"study_design_scores_gemma":[0.00027882878,0.0013926238,0.012450578,0.00012491657,0.00011262994,0.00023373103,0.00090432155,0.8019961,0.016035868,0.15484825,0.011521186,0.000100952646],"about_ca_topic_score_codex":0.0009968195,"about_ca_topic_score_gemma":0.0011406515,"teacher_disagreement_score":0.0032876784,"about_ca_system_score_codex":0.00040072738,"about_ca_system_score_gemma":0.00051619427,"threshold_uncertainty_score":0.010998368},"labels":[],"label_agreement":null},{"id":"W2735649811","doi":"10.1007/s10994-017-5657-1","title":"Generalized exploration in policy search","year":2017,"lang":"en","type":"article","venue":"Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Seventh Framework Programme; Technische Universität Darmstadt","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Entropy (arrow of time); Policy learning; Machine learning; Mathematical optimization; Mathematics","score_opus":0.05004398310810728,"score_gpt":0.33199694268441793,"score_spread":0.28195295957631067,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2735649811","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037790164,0.003500808,0.94850665,0.0016443392,0.00018511902,0.000039637114,0.00007793396,0.0002466478,0.008008659],"genre_scores_gemma":[0.8800489,0.0016316405,0.108621806,0.0003838997,0.0002619137,0.00022231974,0.00014303725,0.00018591779,0.008500661],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983747,0.0011092499,0.000054691172,0.00017740029,0.00018081485,0.00010315369],"domain_scores_gemma":[0.9935149,0.005430716,0.00026674004,0.0003400532,0.00023477199,0.00021285503],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028500005,0.0011298846,0.0025628388,0.0009920279,0.00072784995,0.0018967062,0.0015392064,0.0024668374,0.0038060017],"category_scores_gemma":[0.014741787,0.0009859467,0.0009160815,0.0014635406,0.0042482587,0.0041947816,0.0030933968,0.0028851577,0.00031840062],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011696291,0.000031922773,0.0004452133,0.0001495599,0.00007603399,0.00007792293,0.00013407727,0.5573142,0.00024318208,0.4182181,0.0016937219,0.021499109],"study_design_scores_gemma":[0.000024825156,0.000019550263,0.00006201423,0.000018532639,0.000009647194,0.000012115002,0.000012359383,0.6827138,0.00006106619,0.3165457,0.00051215856,0.000008205477],"about_ca_topic_score_codex":0.0043222373,"about_ca_topic_score_gemma":0.003214812,"teacher_disagreement_score":0.0043222373,"about_ca_system_score_codex":0.0016424874,"about_ca_system_score_gemma":0.001498478,"threshold_uncertainty_score":0.015072405},"labels":[],"label_agreement":null},{"id":"W2737821837","doi":"10.1109/icra.2017.7989686","title":"Adapting learned robotics behaviours through policy adjustment","year":2017,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Robot; Computer science; Inverted pendulum; Task (project management); Artificial intelligence; Robotics; Action (physics); State space; Physical system; Mobile robot; Control theory (sociology); Control (management); Engineering; Mathematics","score_opus":0.07174920135323527,"score_gpt":0.3402462901202425,"score_spread":0.26849708876700723,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2737821837","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035208177,0.000094112445,0.9611998,0.0001727724,0.000042852593,0.000079534024,0.000017493949,0.00095859566,0.0022266218],"genre_scores_gemma":[0.8973403,0.00009127864,0.10028204,0.0001887683,0.000031541746,0.0001768445,0.00004645157,0.00011951345,0.0017233564],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992169,0.00023895282,0.00004663028,0.00020986878,0.00020008217,0.00008753215],"domain_scores_gemma":[0.9978399,0.001047791,0.00034337188,0.00043201025,0.00023915117,0.00009779315],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012501392,0.0009294722,0.00069723197,0.00045946025,0.0003228299,0.00065219897,0.0014444104,0.00095635443,0.0016167115],"category_scores_gemma":[0.0073347846,0.0005260698,0.00052171113,0.00030321814,0.0013552326,0.00101404,0.0011431773,0.0015846096,0.00041243213],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007171124,0.00011934293,0.0012064814,0.000055644832,0.000059041973,0.0000855354,0.00012995412,0.930478,0.0064924094,0.0061127706,0.00050031755,0.054688863],"study_design_scores_gemma":[0.00001461253,0.00004291062,0.0001400387,0.0000053342583,0.0000071919662,0.000018913592,0.000009808209,0.9936162,0.0013608315,0.0043190625,0.0004568775,0.000008203424],"about_ca_topic_score_codex":0.0025105271,"about_ca_topic_score_gemma":0.0019500239,"teacher_disagreement_score":0.0025105271,"about_ca_system_score_codex":0.0007536882,"about_ca_system_score_gemma":0.00086744566,"threshold_uncertainty_score":0.006611407},"labels":[],"label_agreement":null},{"id":"W2738109916","doi":"10.1109/adconip.2017.7983780","title":"Deep reinforcement learning approaches for process control","year":2017,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":137,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Reinforcement learning; Computer science; Process (computing); Controller (irrigation); Artificial neural network; Process control; Control (management); Artificial intelligence; Control system; Control engineering; Function (biology); Deep learning; Control theory (sociology); MIMO; Engineering; Channel (broadcasting)","score_opus":0.049513320306949236,"score_gpt":0.28178940841336486,"score_spread":0.23227608810641562,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2738109916","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0058769374,0.0009916968,0.98955005,0.0003035294,0.00006160966,0.000021630323,0.0000322404,0.0002486222,0.0029136334],"genre_scores_gemma":[0.8685772,0.0012049692,0.12349675,0.0002414217,0.000109706736,0.00015781538,0.00012077758,0.000087252636,0.006004105],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996094,0.00011255759,0.0000238435,0.00007066301,0.00012432427,0.000059268277],"domain_scores_gemma":[0.999151,0.00045915984,0.00008982162,0.00006404134,0.0001947512,0.000041285508],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011382614,0.0010014719,0.00097094226,0.00038575567,0.0002876362,0.00080515054,0.0011169555,0.0009863629,0.0025386757],"category_scores_gemma":[0.0023450346,0.0004066324,0.000503989,0.0004007075,0.0009003193,0.00096053357,0.00095864746,0.00197003,0.0003379282],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000024139457,0.000022683693,0.00018168817,0.000057703248,0.00002412811,0.000025679563,0.000021368107,0.9565096,0.0006054741,0.019674331,0.00042859785,0.022424644],"study_design_scores_gemma":[0.0000041704398,0.0000108583,0.000031220487,0.0000051535553,0.0000027419803,0.000003419219,0.0000018555522,0.9905855,0.00019054902,0.00874764,0.00041416948,0.0000026515654],"about_ca_topic_score_codex":0.0065151,"about_ca_topic_score_gemma":0.004658295,"teacher_disagreement_score":0.0065151,"about_ca_system_score_codex":0.0013887461,"about_ca_system_score_gemma":0.0010564015,"threshold_uncertainty_score":0.012954354},"labels":[],"label_agreement":null},{"id":"W2739573821","doi":"10.3390/make1010002","title":"Learning to Teach Reinforcement Learning Agents","year":2017,"lang":"en","type":"article","venue":"Machine Learning and Knowledge Extraction","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Washington State University; U.S. Department of Agriculture; National Aeronautics and Space Administration; National Science Foundation","keywords":"Reinforcement learning; Advice (programming); Heuristics; Statistic; Action (physics); Variance (accounting); Quality (philosophy); Discounting","score_opus":0.02352967150582671,"score_gpt":0.323069848083718,"score_spread":0.29954017657789134,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2739573821","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07430705,0.00027246927,0.91887003,0.00090212235,0.00006875837,0.00012704321,0.00005040497,0.0004694462,0.0049326103],"genre_scores_gemma":[0.9086771,0.00018848703,0.08585128,0.00023048524,0.000052816664,0.00027047767,0.00006334483,0.000041473675,0.0046244315],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991597,0.00043174086,0.00003301474,0.00014790523,0.00012786538,0.000099747995],"domain_scores_gemma":[0.9943039,0.0043503316,0.00044752288,0.00029383224,0.00035531973,0.00024898577],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017652123,0.00088755845,0.0008937666,0.00032805683,0.00030864013,0.0006809475,0.0014117184,0.0012753432,0.0028543046],"category_scores_gemma":[0.014421762,0.00034789165,0.00034414267,0.000332747,0.0011506857,0.0013166218,0.0008191156,0.0017021941,0.00036414998],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001446207,0.0002238285,0.0017628685,0.0001021635,0.00006490502,0.000090612724,0.00016301023,0.9158089,0.0010065057,0.030670678,0.0010324124,0.048929464],"study_design_scores_gemma":[0.000035070192,0.00005955312,0.00011840785,0.0000073207207,0.0000071024656,0.000011608247,0.000011542113,0.9851724,0.00026033598,0.013882348,0.0004296736,0.000004646145],"about_ca_topic_score_codex":0.0030529057,"about_ca_topic_score_gemma":0.0025624693,"teacher_disagreement_score":0.0030529057,"about_ca_system_score_codex":0.0010633427,"about_ca_system_score_gemma":0.0011236003,"threshold_uncertainty_score":0.009548664},"labels":[],"label_agreement":null},{"id":"W2739747865","doi":"10.24963/ijcai.2017/290","title":"Constrained Bayesian Reinforcement Learning via Approximate Linear Programming","year":2017,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Defense Acquisition Program Administration; Korea Advanced Institute of Science and Technology; Agency for Defense Development; Ministry of Science, ICT and Future Planning","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Bayesian probability; Mathematical optimization; Machine learning; Linear programming; Bayesian optimization; State (computer science); Algorithm; Mathematics","score_opus":0.020878168153838497,"score_gpt":0.2726533437933096,"score_spread":0.2517751756394711,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2739747865","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005258357,0.00012102505,0.9928069,0.0001460792,0.0000128974625,0.000028598695,0.00002092451,0.00024772485,0.0013573929],"genre_scores_gemma":[0.7571067,0.0002378352,0.23763637,0.00029584506,0.000053580196,0.00038983714,0.00016641952,0.0001475398,0.0039658067],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99880993,0.000502071,0.00003888959,0.00019645151,0.0003383625,0.00011429109],"domain_scores_gemma":[0.99709857,0.002080781,0.0002643339,0.000157139,0.0002793433,0.00011987098],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015336608,0.0012415644,0.0017000831,0.00047562778,0.0004098463,0.0010830006,0.0014724174,0.001389802,0.0029131575],"category_scores_gemma":[0.00678737,0.0006641957,0.0005069654,0.00055395666,0.001426023,0.0013745243,0.0016484823,0.0020708218,0.0005200035],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000037273287,0.000036341757,0.00020345354,0.000042486765,0.000018186889,0.000030434514,0.000028192244,0.9646149,0.0003111048,0.015929796,0.00051750906,0.018230248],"study_design_scores_gemma":[0.0000065331915,0.000009992384,0.000012997599,0.0000029876805,0.0000015158699,0.000004043203,0.000001798179,0.9925936,0.00007139673,0.0071629873,0.00013061038,0.0000016771711],"about_ca_topic_score_codex":0.0044159605,"about_ca_topic_score_gemma":0.003986261,"teacher_disagreement_score":0.0044159605,"about_ca_system_score_codex":0.0012518371,"about_ca_system_score_gemma":0.0018211553,"threshold_uncertainty_score":0.009745419},"labels":[],"label_agreement":null},{"id":"W2740174381","doi":"10.24963/ijcai.2017/717","title":"Approximate Value Iteration with Temporally Extended Actions (Extended Abstract)","year":2017,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Landmark; Convergence (economics); Computer science; Mathematical optimization; Bellman equation; Value (mathematics); Function (biology); State space; Term (time); Algorithm; Mathematics; Artificial intelligence; Machine learning; Statistics; Economics","score_opus":0.03463366844770536,"score_gpt":0.2920029215859134,"score_spread":0.257369253138208,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2740174381","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008816935,0.00010357523,0.9891423,0.00010191471,0.000024064822,0.000021849151,0.00004238785,0.00016968421,0.0015773266],"genre_scores_gemma":[0.52886814,0.00021388475,0.46677235,0.00011963012,0.000036367048,0.00021285919,0.0001523097,0.0001178235,0.0035066763],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99901104,0.00045851665,0.00006364371,0.00015235378,0.00022788037,0.00008659168],"domain_scores_gemma":[0.9980077,0.0012928955,0.00020820386,0.00022667987,0.00018320525,0.00008130475],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014213821,0.0006289579,0.0006181485,0.0004098403,0.0003395733,0.0009495034,0.00113365,0.00096682156,0.0049335426],"category_scores_gemma":[0.0068947654,0.0003540804,0.00079691864,0.0006440203,0.0015107171,0.0018173489,0.0015162231,0.001497941,0.0005018036],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018846386,0.000030862622,0.00069908926,0.000092680595,0.000041679228,0.00014728396,0.00016633401,0.74888736,0.001854038,0.20224334,0.0009990125,0.04464979],"study_design_scores_gemma":[0.000015926156,0.000038054353,0.000065968656,0.000017413267,0.0000057727934,0.000029571884,0.000012621187,0.91716045,0.0008081662,0.0805403,0.0012967224,0.000008995833],"about_ca_topic_score_codex":0.0029439083,"about_ca_topic_score_gemma":0.0030005504,"teacher_disagreement_score":0.0049335426,"about_ca_system_score_codex":0.00074080774,"about_ca_system_score_gemma":0.0009157991,"threshold_uncertainty_score":0.016504347},"labels":[],"label_agreement":null},{"id":"W2744625767","doi":"10.48550/arxiv.1708.01298","title":"Effective sketching methods for value function approximation","year":2017,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Sketch; Computer science; Reinforcement learning; Variety (cybernetics); Function (biology); Matrix (chemical analysis); Coding (social sciences); Artificial intelligence; Machine learning; Algorithm; Theoretical computer science; Mathematics","score_opus":0.08423265039614783,"score_gpt":0.26367145793553604,"score_spread":0.1794388075393882,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2744625767","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0019506718,0.00034173508,0.9963619,0.00011520263,0.00002374438,0.00002381448,0.000028500595,0.0002444547,0.000909937],"genre_scores_gemma":[0.2434074,0.0012758914,0.75034815,0.0001779662,0.00010918197,0.00036193428,0.00026505408,0.00033632465,0.003718089],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99850637,0.0006637602,0.000099477074,0.00021798733,0.00043427598,0.000078146906],"domain_scores_gemma":[0.9914034,0.006282633,0.00041712733,0.0011270067,0.0005629616,0.00020703762],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003187149,0.0013087484,0.0013767666,0.0010884778,0.00047651722,0.0017910261,0.001521355,0.001788523,0.0062714084],"category_scores_gemma":[0.022119489,0.00081617053,0.00084229273,0.0010724437,0.0017843549,0.003137549,0.0026969765,0.0032076824,0.001428513],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000103410595,0.000058629215,0.000723782,0.0002910732,0.000058059948,0.00006601353,0.00018566518,0.6436613,0.0024733967,0.18261728,0.0029082918,0.16685304],"study_design_scores_gemma":[0.000018442932,0.00002365237,0.000044117452,0.000028977445,0.0000055694395,0.000020387477,0.000011800165,0.9363264,0.0005581991,0.061358914,0.0015945083,0.000009060081],"about_ca_topic_score_codex":0.0018928745,"about_ca_topic_score_gemma":0.0019391506,"teacher_disagreement_score":0.0062714084,"about_ca_system_score_codex":0.0011251186,"about_ca_system_score_gemma":0.00110851,"threshold_uncertainty_score":0.02098},"labels":[],"label_agreement":null},{"id":"W2750726423","doi":"","title":"Natural Value Approximators: Learning when to Trust Past Estimates","year":2017,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Observability; Reinforcement learning; Bellman equation; Classification of discontinuities; Interpolation (computer graphics); Computer science; Value (mathematics); Artificial neural network; Discontinuity (linguistics); Function approximation; Inductive bias; Artificial intelligence; Function (biology); Mathematical optimization; Algorithm; Machine learning; Mathematics; Applied mathematics; Image (mathematics); Multi-task learning","score_opus":0.01702164821844596,"score_gpt":0.264931605515689,"score_spread":0.24790995729724305,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2750726423","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02005196,0.00022564334,0.97747624,0.00029794525,0.00005856693,0.00005837544,0.000035941575,0.00036995253,0.0014255047],"genre_scores_gemma":[0.7244484,0.00023957608,0.27097338,0.00027804953,0.00011790712,0.0002940262,0.00012583267,0.0001783541,0.003344543],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99873155,0.00043563382,0.000111735324,0.00033369716,0.00025951708,0.00012792854],"domain_scores_gemma":[0.990567,0.006690543,0.0007687417,0.00079596293,0.00084816647,0.000329506],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039599864,0.0010228028,0.0013298645,0.0006717373,0.00043715956,0.0012838929,0.0023752574,0.0022743747,0.0036491062],"category_scores_gemma":[0.026426772,0.00081203185,0.00051111827,0.00046011704,0.0016176414,0.00390638,0.0022573704,0.0033676785,0.00067321974],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006327587,0.00026381636,0.0033317644,0.0002714798,0.00012816921,0.00022392473,0.00046679488,0.63512546,0.0053490163,0.08369867,0.0048460686,0.26566216],"study_design_scores_gemma":[0.000021832851,0.00004590535,0.000083616,0.000024696743,0.0000071737745,0.000024003197,0.000015412512,0.98408246,0.0009631889,0.014262846,0.0004590279,0.000009778083],"about_ca_topic_score_codex":0.0018920457,"about_ca_topic_score_gemma":0.002502193,"teacher_disagreement_score":0.0039599864,"about_ca_system_score_codex":0.00086709816,"about_ca_system_score_gemma":0.0012569693,"threshold_uncertainty_score":0.020942628},"labels":[],"label_agreement":null},{"id":"W2753431380","doi":"","title":"Second-order Optimization for Deep Reinforcement Learning using Kronecker-factored Approximation","year":2017,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Kronecker delta; Reinforcement learning; Computer science; Scalability; Curvature; Trust region; Mathematical optimization; Artificial intelligence; Mathematics","score_opus":0.0371741992360738,"score_gpt":0.28445717157899536,"score_spread":0.24728297234292157,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2753431380","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0040705763,0.0001350124,0.99444985,0.000073521274,0.000025562094,0.000016351363,0.000016750766,0.00034596614,0.00086649845],"genre_scores_gemma":[0.6943551,0.00030579965,0.29762527,0.00020095467,0.00005755537,0.00020841567,0.000180721,0.00045149692,0.006614653],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995004,0.00017095903,0.000027670387,0.00009431076,0.00014915677,0.00005753638],"domain_scores_gemma":[0.9987815,0.0006576595,0.00010250266,0.00013260428,0.0002481574,0.000077586024],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015268205,0.0011904303,0.0012005178,0.0004479228,0.00027701783,0.00091525784,0.0010395824,0.0011787686,0.0028959035],"category_scores_gemma":[0.0045821173,0.00051869464,0.00066440966,0.00038376672,0.0012099405,0.0011553207,0.0010035158,0.0019517636,0.0007713178],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000045300465,0.000020788619,0.00028697366,0.00003909355,0.000024871777,0.000033672794,0.000029187268,0.96668386,0.0011843621,0.012306958,0.0007803082,0.018564703],"study_design_scores_gemma":[0.0000021107917,0.000006874132,0.000010626962,0.0000018091133,0.0000010415558,0.0000028982095,0.0000010060402,0.99797815,0.00014232377,0.0017261109,0.00012569208,0.0000014033762],"about_ca_topic_score_codex":0.006630737,"about_ca_topic_score_gemma":0.005700902,"teacher_disagreement_score":0.006630737,"about_ca_system_score_codex":0.0015345352,"about_ca_system_score_gemma":0.001590851,"threshold_uncertainty_score":0.013184309},"labels":[],"label_agreement":null},{"id":"W2754203286","doi":"10.1609/aaai.v32i1.11831","title":"When Waiting Is Not an Option: Learning Options With a Deliberation Cost","year":2018,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada; Institut de Valorisation des Données","keywords":"Deliberation; Interpretability; Bounded rationality; Computer science; Rationality; Management science; Risk analysis (engineering); Artificial intelligence; Epistemology; Economics; Business; Political science","score_opus":0.12388612333898766,"score_gpt":0.3176455879313191,"score_spread":0.19375946459233143,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2754203286","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14910853,0.00035422522,0.8421383,0.0017867289,0.00006747895,0.0000872173,0.000119173834,0.00059664395,0.0057417606],"genre_scores_gemma":[0.8886179,0.00016922333,0.10798753,0.00024779513,0.000034056644,0.000115434515,0.00012265859,0.00010662609,0.0025987797],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99833196,0.0007454714,0.00008714649,0.00041169822,0.0002340416,0.0001896312],"domain_scores_gemma":[0.9879401,0.009330926,0.0008614725,0.00079553836,0.00044600086,0.0006259366],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032545014,0.00076633255,0.0009888601,0.00037153275,0.0005243709,0.0013867174,0.0016237667,0.0018918565,0.0034410655],"category_scores_gemma":[0.022945076,0.00056125,0.00063092797,0.0003994105,0.0024267929,0.005068857,0.0025801864,0.0031698279,0.00038040476],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012382229,0.00023888843,0.004165752,0.00023506486,0.00013931542,0.00043532668,0.00059534766,0.69575965,0.002742938,0.19205463,0.0021025152,0.100292236],"study_design_scores_gemma":[0.00007075149,0.00008455709,0.00029178985,0.000025610056,0.000022948474,0.000050448456,0.000056822995,0.821645,0.0011096803,0.17594068,0.00067855493,0.000023038601],"about_ca_topic_score_codex":0.0024021433,"about_ca_topic_score_gemma":0.002464376,"teacher_disagreement_score":0.0034410655,"about_ca_system_score_codex":0.0010617786,"about_ca_system_score_gemma":0.0016435824,"threshold_uncertainty_score":0.017211676},"labels":[],"label_agreement":null},{"id":"W2754517384","doi":"10.1609/aaai.v32i1.11694","title":"Deep Reinforcement Learning That Matters","year":2018,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1504,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Open Philanthropy Project; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Computer science; Standardization; Benchmark (surveying); Artificial intelligence; Field (mathematics); Variance (accounting); Deep learning; Machine learning; Data science; Risk analysis (engineering); Mathematics","score_opus":0.07088540185984533,"score_gpt":0.28886615983652203,"score_spread":0.2179807579766767,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2754517384","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09335494,0.05890464,0.38759974,0.34703887,0.014838288,0.00018434964,0.002536151,0.002732888,0.09281014],"genre_scores_gemma":[0.8573246,0.0157044,0.060084257,0.039400376,0.004706854,0.00025642352,0.0010616577,0.0018389026,0.019622583],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9907906,0.0042127613,0.00045221808,0.002325859,0.00190102,0.00031756554],"domain_scores_gemma":[0.9601553,0.02693311,0.0022412387,0.0045945896,0.004792576,0.0012832566],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016862407,0.00068971736,0.0012376839,0.0004689479,0.0009112101,0.0047461004,0.0016060637,0.0024842715,0.008552005],"category_scores_gemma":[0.09086421,0.0003431441,0.00038626092,0.00061897584,0.0040282286,0.008155254,0.0017668633,0.00595044,0.0024463676],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006528513,0.00031010716,0.013748026,0.0013081455,0.000562492,0.00016988144,0.000865069,0.026963344,0.0037342282,0.42260382,0.10344904,0.425633],"study_design_scores_gemma":[0.00014555949,0.0004001205,0.008300369,0.0011953334,0.0001573027,0.00026730634,0.00071666844,0.06270193,0.00626424,0.7822539,0.13747412,0.00012321024],"about_ca_topic_score_codex":0.002344082,"about_ca_topic_score_gemma":0.0018793152,"teacher_disagreement_score":0.016862407,"about_ca_system_score_codex":0.0020131557,"about_ca_system_score_gemma":0.002038659,"threshold_uncertainty_score":0.089178026},"labels":[],"label_agreement":null},{"id":"W2757609746","doi":"10.1609/aaai.v32i1.11775","title":"OptionGAN: Learning Joint Reward-Policy Options Using Generative Adversarial Inverse Reinforcement Learning","year":2018,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Open Philanthropy Project; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Leverage (statistics); Computer science; Adversarial system; Generative grammar; Artificial intelligence; Function (biology); Set (abstract data type); Machine learning","score_opus":0.13871062076476812,"score_gpt":0.32905867573710673,"score_spread":0.19034805497233862,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2757609746","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016302442,0.00027609224,0.9794881,0.00030884348,0.000047481713,0.00006520789,0.00009329027,0.0009759432,0.0024426673],"genre_scores_gemma":[0.8183374,0.0002460204,0.1740828,0.00053379085,0.00006389813,0.00035953856,0.00036936268,0.0002755499,0.005731655],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99946576,0.00021353757,0.000019348785,0.000114220515,0.00012686613,0.000060313578],"domain_scores_gemma":[0.9985569,0.0010526533,0.00009522172,0.00012870603,0.000091278955,0.000075193224],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014739858,0.0011687547,0.0010657208,0.0004470699,0.00025928157,0.00077168655,0.0014934998,0.0017119666,0.003541623],"category_scores_gemma":[0.004755085,0.0006989657,0.0007184935,0.00034174777,0.0014240443,0.0015015954,0.0018791917,0.0024085727,0.0005879542],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007669878,0.000039191953,0.000522191,0.000049198807,0.000045947894,0.00008145621,0.000039616512,0.95446,0.0010786182,0.014614457,0.0012965695,0.027695961],"study_design_scores_gemma":[0.00000723207,0.000013279517,0.000036328995,0.000005370765,0.0000027492993,0.000011421912,0.000001933193,0.9921956,0.00026216713,0.0072427033,0.00021669453,0.0000045544284],"about_ca_topic_score_codex":0.002520568,"about_ca_topic_score_gemma":0.002930377,"teacher_disagreement_score":0.003541623,"about_ca_system_score_codex":0.00077205914,"about_ca_system_score_gemma":0.0010676719,"threshold_uncertainty_score":0.011847913},"labels":[],"label_agreement":null},{"id":"W2759254424","doi":"10.1609/aiide.v13i1.12926","title":"Improvised Theatre Alongside Artificial Intelligences","year":2017,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Improvisation; Conversation; Computer science; Performing arts; Natural (archaeology); Narrative; Artificial intelligence; Visual arts; Linguistics; Psychology; Communication; Art","score_opus":0.05173471714296771,"score_gpt":0.2952074338759871,"score_spread":0.24347271673301937,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2759254424","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.446924,0.002493443,0.29665717,0.006236961,0.001497863,0.00090748706,0.00084910926,0.00399282,0.24044125],"genre_scores_gemma":[0.87742466,0.000535924,0.0840217,0.0008590634,0.0002144625,0.00043421457,0.0006146954,0.0004916555,0.035403673],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9952284,0.0026492977,0.00018889751,0.000607757,0.0010189582,0.00030666066],"domain_scores_gemma":[0.9955675,0.002239514,0.00021616623,0.0009110551,0.00050159666,0.00056431995],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029888179,0.0006296725,0.0004934554,0.00072468835,0.002179684,0.004943687,0.0019475104,0.0015155915,0.013195853],"category_scores_gemma":[0.0091245985,0.00022512273,0.0005554768,0.00035441475,0.0065186196,0.004411263,0.0065053725,0.002038948,0.002302463],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0032025664,0.0013513217,0.006710233,0.003172088,0.00023599912,0.0021995609,0.08159106,0.019512156,0.14362119,0.28283566,0.029423974,0.4261442],"study_design_scores_gemma":[0.00023980676,0.002391017,0.009926665,0.0007457086,0.00010310086,0.002930284,0.029421171,0.028661279,0.06519142,0.0833933,0.77664846,0.0003477959],"about_ca_topic_score_codex":0.0009543879,"about_ca_topic_score_gemma":0.0010292189,"teacher_disagreement_score":0.013195853,"about_ca_system_score_codex":0.0012913918,"about_ca_system_score_gemma":0.00073851336,"threshold_uncertainty_score":0.04414451},"labels":[],"label_agreement":null},{"id":"W2760276144","doi":"10.1609/icaps.v27i1.13803","title":"Analytic Decision Analysis via Symbolic Dynamic Programming for Parameterized Hybrid MDPs","year":2017,"lang":"en","type":"article","venue":"Proceedings of the International Conference on Automated Planning and Scheduling","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Parameterized complexity; Mathematical optimization; Computer science; Leverage (statistics); Discretization; Nonlinear programming; Convex optimization; Dynamic programming; Scalability; Piecewise; Optimization problem; Nonlinear system; Mathematics; Regular polygon; Algorithm; Artificial intelligence","score_opus":0.03706048564634457,"score_gpt":0.3302437727257557,"score_spread":0.29318328707941116,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2760276144","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004916124,0.00022101733,0.99078196,0.0003016314,0.000018386248,0.00004386415,0.000079145684,0.00012831429,0.003509509],"genre_scores_gemma":[0.6025982,0.0010339406,0.3896455,0.00023133984,0.00009071044,0.0008320038,0.0003867902,0.00025654575,0.004924953],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988096,0.0005609846,0.000057364457,0.00015003045,0.00030861262,0.000113368114],"domain_scores_gemma":[0.99365175,0.0053240787,0.00037856918,0.00017773775,0.00033530034,0.000132539],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026924505,0.001603412,0.0015408281,0.0011081863,0.00068882265,0.0023016504,0.0012492315,0.0014173749,0.00381734],"category_scores_gemma":[0.010073344,0.00086273643,0.0013837893,0.0010973169,0.002366537,0.0015457725,0.0024328944,0.002495963,0.00029469936],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000008898623,0.000007870507,0.000119726756,0.000037010403,0.000012896955,0.000028226386,0.00002879675,0.9524428,0.00011176387,0.04380076,0.00018654556,0.0032146627],"study_design_scores_gemma":[0.000003284943,0.0000027805502,0.0000101631995,0.0000058104665,0.0000019688946,0.0000025811023,0.000005491369,0.97706085,0.00004551621,0.022649815,0.00020988597,0.0000018461968],"about_ca_topic_score_codex":0.009015997,"about_ca_topic_score_gemma":0.0073889312,"teacher_disagreement_score":0.009015997,"about_ca_system_score_codex":0.0031495672,"about_ca_system_score_gemma":0.0032323461,"threshold_uncertainty_score":0.022851825},"labels":[],"label_agreement":null},{"id":"W2767354950","doi":"10.1609/aaai.v32i1.11740","title":"Learning With Options That Terminate Off-Policy","year":2018,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Dilemma; Flexibility (engineering); Computer science; Odds; Decoupling (probability); Set (abstract data type); Quality (philosophy); Task (project management); Reinforcement learning; Mathematical optimization; Artificial intelligence; Economics; Mathematics; Machine learning","score_opus":0.08436074975531656,"score_gpt":0.3127829237718858,"score_spread":0.22842217401656928,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2767354950","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06901426,0.00018705128,0.9232509,0.0007430632,0.00007037116,0.00011731634,0.0000782988,0.000742127,0.005796607],"genre_scores_gemma":[0.76441306,0.00017071956,0.22764322,0.0004970367,0.000067374975,0.00032041653,0.00027261974,0.00028108538,0.0063344445],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986358,0.00049656635,0.00008892025,0.00035187072,0.00023363665,0.00019306528],"domain_scores_gemma":[0.99147403,0.005993637,0.00058259815,0.0010307425,0.0005034529,0.0004155097],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033330065,0.001056482,0.0011863428,0.00046341808,0.0005411088,0.0014104465,0.0018330399,0.00239384,0.0040702596],"category_scores_gemma":[0.019831218,0.000487298,0.0008130764,0.00041456186,0.00218879,0.0034985463,0.0022799617,0.003676224,0.0006277933],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049189,0.00028873817,0.0050683585,0.00018423988,0.000076136785,0.00026079945,0.0003485868,0.65198326,0.0032308288,0.19594988,0.002983818,0.1391334],"study_design_scores_gemma":[0.000033975983,0.00007659111,0.00013990235,0.000030653086,0.000010221813,0.000030694682,0.000021487112,0.92471415,0.0011472637,0.0730402,0.0007447589,0.000010108623],"about_ca_topic_score_codex":0.0012242688,"about_ca_topic_score_gemma":0.0015710898,"teacher_disagreement_score":0.0040702596,"about_ca_system_score_codex":0.0013002915,"about_ca_system_score_gemma":0.0019317458,"threshold_uncertainty_score":0.017626822},"labels":[],"label_agreement":null},{"id":"W2771302359","doi":"10.3389/fnbot.2019.00052","title":"A Novel Model for Arbitration Between Planning and Habitual Control Systems","year":2019,"lang":"en","type":"preprint","venue":"Frontiers in Neurorobotics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Action selection; Internal model; Reinforcement learning; Task (project management); Control (management); Action (physics); Artificial intelligence; Kinematics; A priori and a posteriori; Machine learning; Engineering","score_opus":0.042683427339276145,"score_gpt":0.26828128658745193,"score_spread":0.22559785924817577,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2771302359","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033339355,0.0002902616,0.95330065,0.0004976489,0.00010221756,0.00005451265,0.00019136733,0.0012024404,0.011021571],"genre_scores_gemma":[0.92578834,0.00024939657,0.059871018,0.00015416587,0.000058937996,0.00023437108,0.00017182248,0.000097314594,0.013374589],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99972576,0.000042482297,0.000014685578,0.00010480532,0.000055716544,0.000056537738],"domain_scores_gemma":[0.99968004,0.00010590732,0.00006128341,0.000053736676,0.000060084847,0.000039030147],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00040582614,0.0005874878,0.0006264586,0.00024356449,0.00035124467,0.0010663754,0.0016264708,0.0010260566,0.004929658],"category_scores_gemma":[0.0009435276,0.0003987067,0.0006606017,0.00023642079,0.0009493337,0.00131944,0.0011048955,0.0016552025,0.00065361714],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014067133,0.00007723355,0.00091747375,0.00010114749,0.00007421814,0.00024998956,0.00014960811,0.86216325,0.010087853,0.09533268,0.001852757,0.028853094],"study_design_scores_gemma":[0.000013981436,0.000029527904,0.00010518027,0.0000033911379,0.000008364564,0.000030214911,0.0000034053323,0.9857517,0.0005074763,0.01247561,0.0010652535,0.0000058618866],"about_ca_topic_score_codex":0.0031886192,"about_ca_topic_score_gemma":0.0029339471,"teacher_disagreement_score":0.004929658,"about_ca_system_score_codex":0.00072891504,"about_ca_system_score_gemma":0.0010531709,"threshold_uncertainty_score":0.016491354},"labels":[],"label_agreement":null},{"id":"W2783392051","doi":"10.1609/aaai.v32i1.12095","title":"Planning With Pixels in (Almost) Real Time","year":2018,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"University of Alberta","keywords":"Computer science; Pixel; Artificial intelligence; State (computer science); Computer vision; Machine learning; Algorithm","score_opus":0.07757191859625165,"score_gpt":0.3079997685792846,"score_spread":0.23042784998303298,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2783392051","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.077128656,0.0001726409,0.90315455,0.00030336433,0.00007766337,0.00011336545,0.00016709887,0.006506564,0.01237603],"genre_scores_gemma":[0.62552726,0.00009874673,0.3684592,0.00012795802,0.000019221388,0.00013255802,0.0002782939,0.00036144204,0.004995363],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999496,0.00013412462,0.000030070898,0.00016040388,0.00012465894,0.00005482429],"domain_scores_gemma":[0.99904436,0.0004933947,0.000079751684,0.00023809115,0.00009166905,0.00005269455],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00071655907,0.0007006016,0.00056239485,0.00029028094,0.00030185617,0.00095843186,0.001074006,0.0006768594,0.0072928183],"category_scores_gemma":[0.0027391007,0.00033075854,0.00032471187,0.0003170998,0.00093839807,0.0015003663,0.0011119996,0.00087876176,0.0011198347],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010990533,0.00023354724,0.0016486167,0.00017038005,0.00007787195,0.00021328837,0.00032844057,0.5075651,0.020750131,0.03183742,0.0067487527,0.42932746],"study_design_scores_gemma":[0.000068683985,0.000110513356,0.00034202432,0.00001234784,0.00001653763,0.000045717938,0.000055831537,0.9609364,0.010081495,0.025207864,0.003110301,0.000012306248],"about_ca_topic_score_codex":0.0043595973,"about_ca_topic_score_gemma":0.005884976,"teacher_disagreement_score":0.0072928183,"about_ca_system_score_codex":0.0005364275,"about_ca_system_score_gemma":0.0008671292,"threshold_uncertainty_score":0.024396896},"labels":[],"label_agreement":null},{"id":"W2783447922","doi":"10.1109/allerton.2017.8262843","title":"Transition-based versus state-based reward functions for MDPs with Value-at-Risk","year":2017,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Markov decision process; Reinforcement learning; Markov process; Function (biology); Bellman equation; Action (physics); Transformation (genetics); Computer science; Markov chain; Mathematical optimization; State (computer science); Markov model; Mathematics; Artificial intelligence; Machine learning; Statistics; Algorithm","score_opus":0.02514385116644622,"score_gpt":0.2594024720053066,"score_spread":0.23425862083886037,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2783447922","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010859911,0.0004922769,0.9874126,0.00024350148,0.000025844656,0.000031993488,0.000024751196,0.000086066495,0.0008231121],"genre_scores_gemma":[0.8121806,0.001155874,0.18274541,0.00020444556,0.00009069709,0.00029077366,0.00016069507,0.00015341441,0.0030180395],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.996752,0.0019107927,0.00017923319,0.00046645544,0.00043531824,0.00025623824],"domain_scores_gemma":[0.9841248,0.0136469975,0.0008373839,0.00036869495,0.0007030291,0.0003189677],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006621967,0.0015743731,0.0019513404,0.000857376,0.00046212427,0.0016030136,0.0012756657,0.0016660112,0.0019359943],"category_scores_gemma":[0.020236058,0.0006558642,0.0011366843,0.00073571096,0.0023199585,0.003784661,0.0020193723,0.0026621001,0.00026999402],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000812123,0.000043262327,0.00043134103,0.000106195024,0.000035921494,0.00004248176,0.00007196165,0.9139067,0.00040334905,0.06908907,0.00023291392,0.015555588],"study_design_scores_gemma":[0.0000061361866,0.000026963226,0.000062969586,0.000011140579,0.0000071970226,0.000008489058,0.000006295256,0.97815835,0.00015328184,0.021418164,0.00013330445,0.000007716471],"about_ca_topic_score_codex":0.0028797197,"about_ca_topic_score_gemma":0.0015683569,"teacher_disagreement_score":0.006621967,"about_ca_system_score_codex":0.0023034818,"about_ca_system_score_gemma":0.001935436,"threshold_uncertainty_score":0.03502077},"labels":[],"label_agreement":null},{"id":"W2784831250","doi":"10.1109/icra.2018.8462977","title":"Cross-Domain Transfer in Reinforcement Learning Using Target Apprentice","year":2018,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Air Force Office of Scientific Research; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Computer science; Transfer of learning; Task (project management); Domain (mathematical analysis); Sample (material); Artificial intelligence; Apprenticeship; Negative transfer; Sample complexity; Reuse; Multi-task learning; Machine learning; Policy learning; Engineering; Mathematics","score_opus":0.033226289462589445,"score_gpt":0.2995726163778216,"score_spread":0.2663463269152322,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2784831250","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037622273,0.00018067438,0.95880455,0.00014693922,0.000051898627,0.000103487146,0.000018378118,0.0006244735,0.0024472629],"genre_scores_gemma":[0.8794459,0.00013011851,0.115690246,0.00017577369,0.000040531173,0.00025093873,0.00006803962,0.00011097576,0.0040874146],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99901104,0.0004034581,0.000046667155,0.000273458,0.00017448583,0.00009087903],"domain_scores_gemma":[0.99779934,0.0011209879,0.00014078942,0.0005014181,0.00027107578,0.0001663407],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024369345,0.0009791151,0.0010675448,0.00035665432,0.0003427091,0.00082063436,0.0017519455,0.001250179,0.0031995056],"category_scores_gemma":[0.0073915264,0.0004467555,0.000624573,0.00030718613,0.0014507303,0.002013561,0.00298697,0.0021684007,0.0008134689],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036592424,0.0005480032,0.0018653158,0.00017921253,0.00013073761,0.00026053967,0.00030852342,0.75962734,0.011451388,0.027357848,0.0013269193,0.19657826],"study_design_scores_gemma":[0.000023556517,0.00017387459,0.00022625398,0.000007999634,0.000011888792,0.000041980737,0.000016992757,0.9834041,0.0026540123,0.012703578,0.00072421157,0.000011630822],"about_ca_topic_score_codex":0.0010962855,"about_ca_topic_score_gemma":0.00076751434,"teacher_disagreement_score":0.0031995056,"about_ca_system_score_codex":0.000684366,"about_ca_system_score_gemma":0.0007523918,"threshold_uncertainty_score":0.012887895},"labels":[],"label_agreement":null},{"id":"W2785529341","doi":"10.1609/aaai.v32i1.12115","title":"Learning Robust Options","year":2018,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Robustness (evolution); Reinforcement learning; Computer science; Artificial intelligence; Robust control; Mathematical optimization; Machine learning; Convergence (economics); Artificial neural network; Mathematics; Nonlinear system","score_opus":0.09360344398562903,"score_gpt":0.2975536324951889,"score_spread":0.20395018850955987,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2785529341","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025699694,0.00013671766,0.97071606,0.00021606093,0.000028954184,0.000042555308,0.00007905604,0.0006302125,0.0024507977],"genre_scores_gemma":[0.8287028,0.00014406428,0.1674075,0.00022800986,0.000030415029,0.00017058609,0.00024040417,0.00023246549,0.0028437446],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988667,0.00035553292,0.000058417034,0.00032905355,0.00026918054,0.000121063575],"domain_scores_gemma":[0.99675155,0.0018923191,0.00044095315,0.00042499087,0.00030954118,0.000180664],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021550488,0.001021764,0.00115058,0.00058725575,0.00040227274,0.0011583037,0.001537024,0.0013417635,0.003204647],"category_scores_gemma":[0.010562946,0.00067504885,0.0008078803,0.00036348283,0.001789597,0.0026641237,0.0020979943,0.002191251,0.0005565417],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000109401,0.000045151246,0.0010321708,0.00007340592,0.000062016275,0.000094141964,0.00008016323,0.90635586,0.0021337369,0.045522913,0.0009557165,0.043535404],"study_design_scores_gemma":[0.0000074429586,0.000024785482,0.00005211638,0.000008487976,0.0000041772605,0.000012739152,0.000005834067,0.97297645,0.0005946856,0.026007986,0.00029984958,0.0000054158027],"about_ca_topic_score_codex":0.0016515612,"about_ca_topic_score_gemma":0.0019163232,"teacher_disagreement_score":0.003204647,"about_ca_system_score_codex":0.0011736075,"about_ca_system_score_gemma":0.0014609457,"threshold_uncertainty_score":0.011397123},"labels":[],"label_agreement":null},{"id":"W2785940258","doi":"","title":"An inference-based policy gradient method for learning options","year":2018,"lang":"en","type":"article","venue":"UvA-DARE (University of Amsterdam)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Inference; Abstraction; Scalability; Artificial intelligence; Machine learning; A priori and a posteriori; Formalism (music); Differentiable function","score_opus":0.025741822831532517,"score_gpt":0.29847202001778317,"score_spread":0.27273019718625063,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2785940258","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00421276,0.00018051201,0.9940598,0.00013372835,0.000027235024,0.00005794075,0.000028441851,0.000480716,0.0008189812],"genre_scores_gemma":[0.2948868,0.00025737638,0.69987094,0.00030782924,0.00008071089,0.0004343138,0.00021698786,0.00028368513,0.0036614258],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991097,0.0003966544,0.000047381243,0.00016579102,0.0002075301,0.00007293632],"domain_scores_gemma":[0.9975803,0.0018733003,0.00012854829,0.00009733492,0.00023384731,0.000086712185],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002442363,0.0010633665,0.0014302628,0.0009569292,0.0005461169,0.000980706,0.0017846676,0.0017518237,0.00395948],"category_scores_gemma":[0.0077638454,0.00077615096,0.00075147214,0.0007495132,0.0013622767,0.0017511469,0.001463595,0.0026823299,0.0008276938],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022117032,0.00012417516,0.0008085964,0.00012335436,0.000064867316,0.00009769679,0.00009563921,0.75534004,0.0017324953,0.038295686,0.002948912,0.20014736],"study_design_scores_gemma":[0.000015528638,0.000018693852,0.000030895997,0.000007928235,0.0000035062724,0.000008414873,0.0000029841526,0.9919303,0.00028695358,0.007306059,0.0003839501,0.0000047599297],"about_ca_topic_score_codex":0.0045120195,"about_ca_topic_score_gemma":0.0047292802,"teacher_disagreement_score":0.0045120195,"about_ca_system_score_codex":0.0013495325,"about_ca_system_score_gemma":0.002299175,"threshold_uncertainty_score":0.013245761},"labels":[],"label_agreement":null},{"id":"W2785948534","doi":"","title":"NerveNet: Learning Structured Policy with Graph Neural Networks","year":2018,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":155,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Concatenation (mathematics); Computer science; Transfer of learning; Artificial intelligence; Graph; Benchmarking; Machine learning; Artificial neural network; Control (management); Theoretical computer science; Mathematics","score_opus":0.031554453052495206,"score_gpt":0.32419013458969764,"score_spread":0.29263568153720243,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2785948534","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032740634,0.00034727834,0.9600867,0.00048180955,0.00016867921,0.00007158482,0.00019355922,0.0024932988,0.0034164481],"genre_scores_gemma":[0.7861734,0.00030074967,0.20808122,0.0004578753,0.000076106575,0.00022769622,0.0004892196,0.00025078256,0.0039429544],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997265,0.00008783876,0.000012912957,0.00006914877,0.00006718761,0.000036475172],"domain_scores_gemma":[0.99900466,0.0005756516,0.00010357851,0.00012102638,0.00012694723,0.00006810696],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008886644,0.0007510662,0.0007149902,0.00050936424,0.00027666995,0.0005849849,0.0011917482,0.0011066562,0.0025606735],"category_scores_gemma":[0.0035730852,0.00037503976,0.00040340231,0.0004310197,0.0008421114,0.0012505923,0.0008722362,0.0014437492,0.000490779],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004468929,0.000046543708,0.0003578648,0.000035179804,0.000021903486,0.000034555702,0.00001687485,0.9576289,0.00058873446,0.007076573,0.001427115,0.03272104],"study_design_scores_gemma":[0.0000045075085,0.0000143057205,0.000021747232,0.0000024530282,0.0000015889651,0.000004024357,0.0000012296136,0.99560463,0.00021779373,0.0038965833,0.00022943724,0.0000016861508],"about_ca_topic_score_codex":0.0054361625,"about_ca_topic_score_gemma":0.0068445303,"teacher_disagreement_score":0.0054361625,"about_ca_system_score_codex":0.0008313048,"about_ca_system_score_gemma":0.0013094348,"threshold_uncertainty_score":0.010809004},"labels":[],"label_agreement":null},{"id":"W2786478526","doi":"","title":"Adversarial Policy Gradient for Alternating Markov Games","year":2018,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Artificial neural network; Computer science; Markov decision process; Markov chain; Gradient descent; Mathematical optimization; Temporal difference learning; Artificial intelligence; Markov process; Machine learning; Mathematics; Statistics","score_opus":0.05030579220343445,"score_gpt":0.3677090667664834,"score_spread":0.31740327456304895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2786478526","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02031145,0.0002300542,0.9742187,0.00038562244,0.00006183327,0.000069901085,0.000053189055,0.00033449294,0.0043347436],"genre_scores_gemma":[0.8357289,0.00030651336,0.15341434,0.00035601997,0.00007336227,0.00041076008,0.00017690587,0.00017963897,0.009353611],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99911314,0.00046415703,0.00003379303,0.00014442434,0.00015233185,0.00009210389],"domain_scores_gemma":[0.9965191,0.0027393636,0.00021774339,0.000127071,0.00023333583,0.00016335514],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018283668,0.0012050403,0.0012533559,0.0005399686,0.00039849317,0.0008780608,0.0010631426,0.0012129257,0.0036926093],"category_scores_gemma":[0.007792008,0.000496726,0.000555927,0.00035300135,0.0016717671,0.0011627283,0.0013799334,0.0019036671,0.00045380648],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006494252,0.000031257387,0.00029837684,0.00004999825,0.000019521796,0.000042715306,0.000041912816,0.93521893,0.0004969276,0.05240951,0.00076145766,0.010564385],"study_design_scores_gemma":[0.000005016308,0.0000087153785,0.000018987872,0.0000031197922,0.0000015537679,0.0000036815568,0.0000021622661,0.9887982,0.000090499896,0.010894276,0.00017149361,0.0000022526872],"about_ca_topic_score_codex":0.004118285,"about_ca_topic_score_gemma":0.0033296468,"teacher_disagreement_score":0.004118285,"about_ca_system_score_codex":0.0016766997,"about_ca_system_score_gemma":0.0015499096,"threshold_uncertainty_score":0.012352943},"labels":[],"label_agreement":null},{"id":"W2787477569","doi":"10.1609/aaai.v32i1.11594","title":"PAC Reinforcement Learning With an Imperfect Model","year":2018,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"McGill University","keywords":"Reinforcement learning; Fidelity; Computer science; Reduction (mathematics); Action (physics); Simple (philosophy); Artificial intelligence; Sample (material); Transfer of learning; Imperfect; Sample complexity; State space; Polynomial; Machine learning; Mathematics","score_opus":0.0664821800134268,"score_gpt":0.2958776474047536,"score_spread":0.22939546739132677,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2787477569","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029168203,0.00019484841,0.96531147,0.00061333145,0.00005130838,0.00006492491,0.000091074515,0.00062980124,0.0038750586],"genre_scores_gemma":[0.8438513,0.00016947159,0.14952601,0.0003128973,0.00009374286,0.0002572969,0.00022092134,0.00012045308,0.005447965],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987025,0.00048087482,0.000059424387,0.00032473728,0.00025534784,0.00017704237],"domain_scores_gemma":[0.9915251,0.0061417464,0.0005600767,0.0010777707,0.00037009173,0.00032519005],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021210276,0.0013157048,0.002045392,0.00037255886,0.00053145067,0.0012975743,0.0024507246,0.0017379118,0.0025305094],"category_scores_gemma":[0.011053077,0.00075272063,0.000606444,0.00051832694,0.00222721,0.002906268,0.00189225,0.0037911837,0.00046063573],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016483915,0.00006582003,0.0004742221,0.00008660168,0.000038576283,0.000098401804,0.0000610417,0.9322212,0.00041068392,0.05089907,0.0012330416,0.014246485],"study_design_scores_gemma":[0.00001978225,0.000021989366,0.000034806355,0.0000038406733,0.0000046636246,0.000009713055,0.0000036563692,0.9791605,0.0001514287,0.020388745,0.0001970322,0.0000038542203],"about_ca_topic_score_codex":0.0047953897,"about_ca_topic_score_gemma":0.004883823,"teacher_disagreement_score":0.0047953897,"about_ca_system_score_codex":0.001656089,"about_ca_system_score_gemma":0.0023176232,"threshold_uncertainty_score":0.01201582},"labels":[],"label_agreement":null},{"id":"W2788842776","doi":"10.3389/frobt.2018.00079","title":"Reactive Reinforcement Learning in Asynchronous Environments","year":2018,"lang":"en","type":"article","venue":"Frontiers in Robotics and AI","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates; Canada Research Chairs; DeepMind; Canada Foundation for Innovation; Alberta Machine Intelligence Institute","keywords":"Reinforcement learning; Computer science; Asynchronous communication; Markov decision process; Artificial intelligence; Distributed computing; Machine learning; Markov process; Computer network","score_opus":0.00789269640831681,"score_gpt":0.22459539567674328,"score_spread":0.21670269926842647,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2788842776","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03712673,0.00013638436,0.9588155,0.00016353514,0.00004702676,0.00004055499,0.000020035932,0.00038849766,0.0032616977],"genre_scores_gemma":[0.9217573,0.000122041376,0.07525771,0.00010739426,0.000035912413,0.00011619509,0.000037145444,0.00004502682,0.0025211952],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999158,0.0003244575,0.000040251565,0.00018565972,0.00018973002,0.000101824764],"domain_scores_gemma":[0.99714214,0.0017982402,0.0003488755,0.00024373093,0.0002931747,0.00017385208],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014425424,0.00069185335,0.00065437134,0.00026692415,0.0003878024,0.0007919197,0.0012528909,0.0008089314,0.0014217204],"category_scores_gemma":[0.005119475,0.00029562207,0.0003414223,0.00020713935,0.0010557817,0.0010666553,0.0009863973,0.0012401373,0.00025635425],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001085388,0.0000533443,0.0004558444,0.000044308163,0.000023438892,0.00007792382,0.00006529514,0.94947547,0.0018133033,0.02482207,0.00044085612,0.02261951],"study_design_scores_gemma":[0.000016331023,0.000032006963,0.000050404564,0.00000229071,0.0000032921798,0.000009396283,0.00000523762,0.9910471,0.00035571656,0.008127902,0.0003471261,0.0000032345868],"about_ca_topic_score_codex":0.0023606531,"about_ca_topic_score_gemma":0.0013662284,"teacher_disagreement_score":0.0023606531,"about_ca_system_score_codex":0.0006744492,"about_ca_system_score_gemma":0.0007849872,"threshold_uncertainty_score":0.0076289773},"labels":[],"label_agreement":null},{"id":"W2791039059","doi":"10.1007/978-3-319-77553-1_9","title":"Scaling Tangled Program Graphs to Visual Reinforcement Learning in ViZDoom","year":2018,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Reinforcement learning; Computer science; Task (project management); Artificial intelligence; Frame (networking); Process (computing); Code (set theory); Pixel; Graph; Machine learning; Theoretical computer science; Programming language","score_opus":0.01606880271556596,"score_gpt":0.27940287635503464,"score_spread":0.2633340736394687,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2791039059","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043715794,0.00038982596,0.9453955,0.00023297491,0.000094922405,0.000043613512,0.00008459773,0.0020611263,0.007981549],"genre_scores_gemma":[0.7294443,0.00035137488,0.2597354,0.00013702745,0.000044343727,0.00011576887,0.0001941215,0.0006908024,0.009286866],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998385,0.000053847387,0.0000070705323,0.00004221591,0.00003874744,0.000019707482],"domain_scores_gemma":[0.9994802,0.0002760416,0.000037579215,0.000086401815,0.0000726751,0.000047155663],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003300432,0.00042205743,0.00054356526,0.00040417508,0.0002906814,0.0005775404,0.00078351714,0.00048056492,0.0067831823],"category_scores_gemma":[0.002549964,0.0002570692,0.00031090114,0.0003718267,0.0005277266,0.001302201,0.0011264145,0.0010575529,0.0005541082],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018986061,0.0000977163,0.00040595746,0.00013860289,0.000021613567,0.00008153532,0.00013440363,0.6360936,0.006118827,0.11816068,0.00624686,0.23231027],"study_design_scores_gemma":[0.000008828456,0.000021088215,0.000057784993,0.000008749475,0.0000032755445,0.000009726847,0.000011395067,0.93431145,0.00074747106,0.06356777,0.0012482856,0.0000041395156],"about_ca_topic_score_codex":0.0029367637,"about_ca_topic_score_gemma":0.0040935986,"teacher_disagreement_score":0.0067831823,"about_ca_system_score_codex":0.0006258234,"about_ca_system_score_gemma":0.00044081276,"threshold_uncertainty_score":0.022692025},"labels":[],"label_agreement":null},{"id":"W2791696004","doi":"10.1109/cig.2018.8490448","title":"Automated Curriculum Learning by Rewarding Temporally Rare Events","year":2018,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University","keywords":"Reinforcement learning; Computer science; Event (particle physics); Function (biology); Simple (philosophy); Artificial intelligence; Human–computer interaction; Machine learning; Cognitive psychology; Psychology","score_opus":0.014086455632860425,"score_gpt":0.2715538308190012,"score_spread":0.25746737518614077,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2791696004","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15006915,0.00020349161,0.8425165,0.00033462635,0.000060278915,0.000108926404,0.00007955922,0.0015838463,0.0050436566],"genre_scores_gemma":[0.9201692,0.00008540311,0.07716703,0.000079333964,0.00002278862,0.00009085645,0.00006794911,0.00007935651,0.002238029],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99947137,0.00018036056,0.000029294219,0.00014622908,0.000095425065,0.00007736399],"domain_scores_gemma":[0.9970714,0.0015030113,0.00045139214,0.0004890152,0.00021290436,0.00027224573],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010595743,0.0005743352,0.0006175269,0.00030064103,0.00032365488,0.0007506433,0.001045002,0.00065607514,0.0028264741],"category_scores_gemma":[0.007248548,0.00032166127,0.00031450094,0.00022222832,0.0009035824,0.0013687003,0.001508244,0.0012873134,0.0004697552],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005400452,0.00035455677,0.005675492,0.00020953461,0.000077278535,0.0002167971,0.00022745114,0.72058046,0.02977279,0.04725147,0.0021569242,0.19293718],"study_design_scores_gemma":[0.00004657606,0.000105311,0.00047369007,0.0000113984,0.000011279181,0.000040268187,0.000012893003,0.9744862,0.0060345884,0.017533908,0.0012295424,0.00001440971],"about_ca_topic_score_codex":0.0006613983,"about_ca_topic_score_gemma":0.0012023774,"teacher_disagreement_score":0.0028264741,"about_ca_system_score_codex":0.0005396776,"about_ca_system_score_gemma":0.0007771712,"threshold_uncertainty_score":0.009455502},"labels":[],"label_agreement":null},{"id":"W2794940174","doi":"10.1609/aimag.v39i1.2780","title":"Constructing Temporal Abstractions Autonomously in Reinforcement Learning","year":2018,"lang":"en","type":"article","venue":"AI Magazine","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Computer science; Abstraction; Artificial intelligence; Thread (computing); Architecture; Programming language","score_opus":0.015208194459727754,"score_gpt":0.2693107594008761,"score_spread":0.25410256494114836,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2794940174","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030147891,0.0001681402,0.9665736,0.00030146184,0.000032708405,0.00002725927,0.000025568399,0.0003922803,0.0023310953],"genre_scores_gemma":[0.8272767,0.00024376698,0.16956156,0.00011908169,0.000025790827,0.0001148586,0.00005669069,0.000067720895,0.0025337003],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99957293,0.00016622657,0.00002012528,0.00009345554,0.0000941359,0.000053112326],"domain_scores_gemma":[0.9989016,0.0006443942,0.00012347722,0.00014558164,0.00008865081,0.00009634471],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011839732,0.0005161537,0.000495887,0.00021159403,0.000397307,0.0008488545,0.0010619517,0.0008354679,0.0014596747],"category_scores_gemma":[0.0036286574,0.00046352082,0.0004786377,0.00024108242,0.0017343013,0.0021949918,0.0014601792,0.0019125001,0.00021349837],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016427912,0.0000647896,0.0009768545,0.000089720386,0.00006265267,0.00017433753,0.0003130243,0.7613071,0.005656652,0.18493032,0.0009968576,0.045263406],"study_design_scores_gemma":[0.000018077697,0.00002681528,0.00005950228,0.000007279524,0.000008949527,0.000014965228,0.000011223081,0.93646324,0.00089928403,0.061733257,0.0007495049,0.00000795724],"about_ca_topic_score_codex":0.001988796,"about_ca_topic_score_gemma":0.0024582958,"teacher_disagreement_score":0.001988796,"about_ca_system_score_codex":0.00071623165,"about_ca_system_score_gemma":0.00092093856,"threshold_uncertainty_score":0.0062615275},"labels":[],"label_agreement":null},{"id":"W2794958591","doi":"10.1109/iros.2018.8594242","title":"Accelerating Learning in Constructive Predictive Frameworks with the Successor Representation","year":2018,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Artificial intelligence; Constructive; Successor cardinal; Reinforcement learning; Interdependence; Representation (politics); Machine learning; Process (computing); Task (project management); Robot; Grid; Field (mathematics); Robotics; Engineering","score_opus":0.024046678949333006,"score_gpt":0.2852600510444348,"score_spread":0.2612133720951018,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2794958591","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017466724,0.00020855616,0.97962403,0.00021772283,0.000028782051,0.00003038405,0.00003904434,0.0007091186,0.0016756679],"genre_scores_gemma":[0.68747747,0.00034230176,0.30895,0.00021196099,0.00007674097,0.00018199814,0.0001628312,0.00016628522,0.0024303973],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991831,0.00028929362,0.0000372786,0.00017741344,0.00020649463,0.0001063071],"domain_scores_gemma":[0.99647874,0.0021343322,0.0003118966,0.000567302,0.0003464998,0.00016120484],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024284709,0.00085889775,0.0010780046,0.00085797266,0.00041431197,0.0012876316,0.0022787102,0.0012902613,0.0027632955],"category_scores_gemma":[0.008472736,0.0005115501,0.0008857846,0.0007561647,0.0019444756,0.0036193805,0.0020868788,0.0020879523,0.0005318447],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010506118,0.00009555217,0.0009256602,0.00011656994,0.000049681184,0.00013062912,0.00016435204,0.7919825,0.0023035684,0.107141286,0.0012063563,0.09577879],"study_design_scores_gemma":[0.00001229281,0.000036006815,0.000039833536,0.000009055347,0.000006711494,0.000015926418,0.000005441035,0.96730405,0.0006340982,0.031457677,0.0004731134,0.000005773328],"about_ca_topic_score_codex":0.002829548,"about_ca_topic_score_gemma":0.00311911,"teacher_disagreement_score":0.002829548,"about_ca_system_score_codex":0.0008840165,"about_ca_system_score_gemma":0.0013136087,"threshold_uncertainty_score":0.012843132},"labels":[],"label_agreement":null},{"id":"W2796085096","doi":"10.1007/978-3-319-89656-4_6","title":"Advice-Based Exploration in Model-Based Reinforcement Learning","year":2018,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Centre for Social Innovation; Vector Institute; University of Toronto","funders":"","keywords":"Reinforcement learning; Satisficing; Computer science; Advice (programming); Robustness (evolution); Convergence (economics); Grid; Artificial intelligence; Operations research; Mathematical optimization; Engineering","score_opus":0.026662711587482006,"score_gpt":0.2551386000725917,"score_spread":0.2284758884851097,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2796085096","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011196015,0.0007559029,0.98253685,0.00017959904,0.0000752644,0.000026386975,0.00002229967,0.00030766014,0.0049000853],"genre_scores_gemma":[0.8023399,0.00084504037,0.1877841,0.00012709982,0.00010017078,0.00021522568,0.00008464736,0.00016296332,0.0083408225],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995215,0.00020733032,0.000025382627,0.00007008334,0.0001308916,0.000044818316],"domain_scores_gemma":[0.9985141,0.0011553866,0.00006518456,0.0000951766,0.00011324306,0.000056867244],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009662018,0.00065761665,0.0009562142,0.00028992578,0.00030671645,0.0008501934,0.0013928667,0.0010481202,0.0036555782],"category_scores_gemma":[0.004286531,0.0004889423,0.00050373684,0.00044423935,0.0010243356,0.0011841796,0.0013504075,0.0018270193,0.0004055896],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000114455106,0.000058013087,0.0002563703,0.00013197718,0.0000321198,0.000047607806,0.00008594605,0.8493829,0.0013740638,0.06387893,0.0016201078,0.0830176],"study_design_scores_gemma":[0.000008543843,0.00001682929,0.000021364487,0.000006297943,0.0000033094811,0.0000075627154,0.0000021535634,0.9805722,0.00017954444,0.018864227,0.00031496395,0.0000029045686],"about_ca_topic_score_codex":0.0027488947,"about_ca_topic_score_gemma":0.0022913935,"teacher_disagreement_score":0.0036555782,"about_ca_system_score_codex":0.0007163897,"about_ca_system_score_gemma":0.00068491505,"threshold_uncertainty_score":0.012229145},"labels":[],"label_agreement":null},{"id":"W2796389682","doi":"10.1109/cdc.2018.8619180","title":"Renewal Monte Carlo: Renewal Theory Based Reinforcement Learning","year":2018,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Markov decision process; Monte Carlo method; Estimator; Computer science; Reinforcement learning; Mathematical optimization; Importance sampling; Variance (accounting); Key (lock); Markov process; Mathematics; Artificial intelligence; Statistics; Economics","score_opus":0.0209271306621518,"score_gpt":0.2568127963650557,"score_spread":0.23588566570290392,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2796389682","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0023964928,0.00027947003,0.9925527,0.00023148645,0.00006946839,0.00004600873,0.000045794884,0.001246486,0.0031320567],"genre_scores_gemma":[0.43078887,0.0009411195,0.5544857,0.0005578401,0.00023405772,0.0004287326,0.00039420195,0.00048686404,0.011682645],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9984694,0.0005679768,0.000056940673,0.0002501267,0.00051104947,0.00014442904],"domain_scores_gemma":[0.99728644,0.0017567076,0.00019884293,0.00023579293,0.00035844045,0.0001636728],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020390442,0.00085586024,0.0011502983,0.0007940037,0.0004905066,0.001750836,0.0019314849,0.0014623578,0.006458944],"category_scores_gemma":[0.007852039,0.00053810945,0.0007184631,0.0008724486,0.0011343797,0.0015917335,0.0015723282,0.002277542,0.0014778157],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015346055,0.00015134801,0.0009012585,0.000121999874,0.000059630165,0.00012519855,0.000068629444,0.74000776,0.0012345216,0.09773013,0.006782459,0.1526636],"study_design_scores_gemma":[0.0000141477985,0.000018795448,0.00004860712,0.000010812183,0.0000065982704,0.00002850435,0.0000028117706,0.9789934,0.0004074776,0.01792058,0.0025400491,0.000008104084],"about_ca_topic_score_codex":0.0057195826,"about_ca_topic_score_gemma":0.0041286536,"teacher_disagreement_score":0.006458944,"about_ca_system_score_codex":0.0015385814,"about_ca_system_score_gemma":0.002395795,"threshold_uncertainty_score":0.02160728},"labels":[],"label_agreement":null},{"id":"W2800222226","doi":"10.7939/r3cw3c","title":"A general framework for reducing variance in agent evaluation","year":2010,"lang":"en","type":"article","venue":"University of Alberta Library","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Variance (accounting); Computer science; Risk analysis (engineering); Business","score_opus":0.01671819584397845,"score_gpt":0.2321567703057499,"score_spread":0.21543857446177145,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2800222226","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00073995383,0.00016544363,0.99765164,0.0001374353,0.000021523854,0.00005003281,0.000014817314,0.00008187601,0.0011372223],"genre_scores_gemma":[0.16667286,0.00059006363,0.8273221,0.00043282105,0.000306918,0.0007329733,0.00014153411,0.00029657347,0.003504124],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.981299,0.009015297,0.0009548577,0.0023859346,0.005587408,0.0007574596],"domain_scores_gemma":[0.97911525,0.013507986,0.001516092,0.0025808217,0.0028943608,0.00038543495],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019434473,0.0022569709,0.0024593298,0.0027423694,0.0011518735,0.003970855,0.0037445256,0.0023191872,0.004002336],"category_scores_gemma":[0.04765212,0.0010600312,0.0023895486,0.0017279184,0.003082165,0.004846109,0.004776303,0.0046149422,0.000982901],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011764113,0.00014135272,0.0016598313,0.00028345658,0.00026294574,0.00014261359,0.00036857955,0.30186197,0.0030937933,0.52638125,0.0032419167,0.16244464],"study_design_scores_gemma":[0.000042484495,0.00014920226,0.0005341549,0.00008306669,0.000069722126,0.00009656105,0.00003916409,0.736428,0.0021279203,0.25406617,0.0063092676,0.00005435073],"about_ca_topic_score_codex":0.0024878925,"about_ca_topic_score_gemma":0.0021306826,"teacher_disagreement_score":0.019434473,"about_ca_system_score_codex":0.002720442,"about_ca_system_score_gemma":0.0029391926,"threshold_uncertainty_score":0.10278052},"labels":[],"label_agreement":null},{"id":"W2804948070","doi":"","title":"Using Reward Machines for High-Level Task Specification and Decomposition in Reinforcement Learning","year":2018,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":129,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Task (project management); Decomposition; Artificial intelligence; Human–computer interaction; Engineering","score_opus":0.09333430035172953,"score_gpt":0.3548435333796292,"score_spread":0.2615092330278997,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2804948070","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004756275,0.000041258932,0.9940925,0.00006205506,0.000016190917,0.000030773026,0.000017067112,0.00052994146,0.0004539775],"genre_scores_gemma":[0.54628175,0.00009231898,0.45050892,0.00014927126,0.000033181404,0.0003952037,0.00012237884,0.00028440976,0.0021324188],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99867934,0.0005568902,0.00009434766,0.0002207119,0.00026965622,0.00017914153],"domain_scores_gemma":[0.99584097,0.0029097782,0.0002672012,0.00041070892,0.00037690855,0.00019441485],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025481451,0.0010088326,0.0013803495,0.00048519578,0.0005417572,0.0014187463,0.0015377033,0.001567223,0.00428763],"category_scores_gemma":[0.009726472,0.00089142483,0.0008940281,0.00045099796,0.0013413043,0.0021877664,0.0020535844,0.003628187,0.0008703198],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020834302,0.00015490611,0.0006406398,0.00012367446,0.00004840447,0.00008591618,0.00014443809,0.83960855,0.005107183,0.035829563,0.0012820783,0.116766214],"study_design_scores_gemma":[0.000008479065,0.000015785754,0.000032920485,0.000005340651,0.0000031937263,0.0000042634565,0.0000033063818,0.987394,0.00072068,0.011642279,0.00016538051,0.0000043514237],"about_ca_topic_score_codex":0.0028536967,"about_ca_topic_score_gemma":0.004632657,"teacher_disagreement_score":0.00428763,"about_ca_system_score_codex":0.0010634571,"about_ca_system_score_gemma":0.001747118,"threshold_uncertainty_score":0.01434356},"labels":[],"label_agreement":null},{"id":"W2807787719","doi":"10.65109/mhju3374","title":"Leveraging Observational Learning for Exploration in Bandits","year":2018,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Reinforcement learning; Imitation; Observational learning; Observational study; Function (biology); Artificial intelligence; Multi-agent system; Intelligent agent; Machine learning; Cloning (programming); Psychology; Mathematics","score_opus":0.13077818732367666,"score_gpt":0.3125012498775136,"score_spread":0.18172306255383694,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2807787719","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03227261,0.00070596894,0.96311164,0.0005413322,0.0000507894,0.00005476988,0.00006558836,0.00041980532,0.002777578],"genre_scores_gemma":[0.9348398,0.00061178324,0.0611389,0.00019000217,0.00013128173,0.00026871782,0.00009968708,0.000097886856,0.0026220414],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977532,0.0011288328,0.0001300551,0.00036968954,0.0003644429,0.000253833],"domain_scores_gemma":[0.9849133,0.011528536,0.0014901749,0.001135307,0.00048787036,0.00044489742],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042851632,0.001606095,0.002484645,0.0008236565,0.0008061773,0.002273983,0.002589757,0.0025114026,0.0032561293],"category_scores_gemma":[0.027000261,0.0008402634,0.00087159907,0.0007917439,0.0031367347,0.00359405,0.0036898246,0.0027746721,0.0005857582],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018980971,0.00011129706,0.0018984001,0.00016678852,0.00008749571,0.00016513835,0.00018576214,0.8469006,0.0012612474,0.119853996,0.00064651907,0.028532956],"study_design_scores_gemma":[0.00001636232,0.000033971548,0.00008683664,0.000015260644,0.000008167974,0.000015855805,0.000008205544,0.9477517,0.00014240968,0.051705055,0.00020824173,0.000007879688],"about_ca_topic_score_codex":0.0027662353,"about_ca_topic_score_gemma":0.0025694843,"teacher_disagreement_score":0.0042851632,"about_ca_system_score_codex":0.0013362373,"about_ca_system_score_gemma":0.0014201378,"threshold_uncertainty_score":0.022662342},"labels":[],"label_agreement":null},{"id":"W2808117931","doi":"10.65109/ohiq4340","title":"Faster Policy Adaptation in Environments with Exogeneity: A State Augmentation Approach","year":2018,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Kernel (algebra); State (computer science); Embedding; Subspace topology; Variable (mathematics); State space; Function (biology); Q-learning; Variance (accounting); Artificial intelligence; Algorithm; Mathematics","score_opus":0.022305160159295453,"score_gpt":0.24347652072051407,"score_spread":0.2211713605612186,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2808117931","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014612573,0.0001683471,0.98358065,0.00021733524,0.000037369973,0.0000370868,0.000019710595,0.00057235226,0.00075465493],"genre_scores_gemma":[0.8257242,0.00015520585,0.17201044,0.0002731374,0.000077451616,0.00016576058,0.0000887569,0.000104535306,0.0014004321],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993542,0.00025448189,0.000049034214,0.00016867259,0.000098526136,0.000075029035],"domain_scores_gemma":[0.99619365,0.0025381828,0.00035491708,0.00050637074,0.00028469882,0.00012215493],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023015516,0.0009144843,0.0015339493,0.0005023944,0.0004073989,0.00093803246,0.0014493695,0.0012582211,0.0023068609],"category_scores_gemma":[0.0070229066,0.00067564414,0.00062308775,0.00043944962,0.0013146062,0.0025001408,0.0021959832,0.0022654925,0.0004025856],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021512153,0.00014904854,0.0011040665,0.00007779625,0.00005981816,0.0001157888,0.000168029,0.8792643,0.0037217075,0.016462287,0.0008913021,0.097770706],"study_design_scores_gemma":[0.0000072683283,0.000013891845,0.000032988177,0.0000026696034,0.0000025045833,0.0000073794376,0.0000026173595,0.99683064,0.00024317057,0.0027244918,0.00013000611,0.00000245896],"about_ca_topic_score_codex":0.002672307,"about_ca_topic_score_gemma":0.0020322672,"teacher_disagreement_score":0.002672307,"about_ca_system_score_codex":0.0005804811,"about_ca_system_score_gemma":0.0013443071,"threshold_uncertainty_score":0.012171924},"labels":[],"label_agreement":null},{"id":"W2808315817","doi":"10.65109/ntef7045","title":"Eligibility Traces for Options","year":2018,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Backup; Reinforcement learning; Abstraction; Sampling (signal processing); Tree (set theory); Machine learning; Database","score_opus":0.0352984014728696,"score_gpt":0.33755489885714995,"score_spread":0.30225649738428034,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2808315817","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025919253,0.00011439835,0.9691713,0.0003027991,0.00005942979,0.000086568114,0.00019093514,0.00064592983,0.0035093322],"genre_scores_gemma":[0.72544926,0.00012696283,0.2687194,0.0001906785,0.00003162815,0.0002866822,0.0003182024,0.00020912939,0.004668046],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.999087,0.00031507068,0.00006411429,0.00017998644,0.0002615874,0.000092399365],"domain_scores_gemma":[0.9962998,0.0022587702,0.0002540067,0.0005768128,0.00034440224,0.00026622394],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012391336,0.00052681944,0.0004986811,0.00061309227,0.00061547535,0.0011057404,0.0015831648,0.0009852547,0.007906214],"category_scores_gemma":[0.011662093,0.00034704825,0.0005423318,0.00044164006,0.0016268069,0.004153659,0.0018950088,0.002491866,0.00057233404],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037954864,0.00016726712,0.0015267539,0.00022534671,0.000038673592,0.00028165718,0.00042998203,0.41308632,0.0050189067,0.4522303,0.0029596193,0.123655535],"study_design_scores_gemma":[0.000029577486,0.00006929943,0.000244143,0.000040740277,0.000011378314,0.000048388905,0.000049127302,0.72249323,0.0025229503,0.27083844,0.003631957,0.00002075528],"about_ca_topic_score_codex":0.0028375036,"about_ca_topic_score_gemma":0.004704736,"teacher_disagreement_score":0.007906214,"about_ca_system_score_codex":0.0009130081,"about_ca_system_score_gemma":0.0016232282,"threshold_uncertainty_score":0.026448965},"labels":[],"label_agreement":null},{"id":"W2808386811","doi":"10.65109/fknq6967","title":"Teaching Multiple Tasks to an RL Agent using LTL","year":2018,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":89,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Linear temporal logic; Convergence (economics); Exploit; Temporal logic; State (computer science); Artificial intelligence; Model checking; Theoretical computer science; Programming language","score_opus":0.05619774051327824,"score_gpt":0.32256481769397255,"score_spread":0.2663670771806943,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2808386811","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015189695,0.00004500354,0.981121,0.00019742595,0.00002084342,0.00006565416,0.000024014249,0.0010720844,0.002264343],"genre_scores_gemma":[0.6399556,0.00010427034,0.35611027,0.00019152203,0.000024022738,0.00021623353,0.00008730694,0.00015838022,0.0031523874],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999495,0.00018122882,0.00003467238,0.00011022789,0.00012425019,0.00005456463],"domain_scores_gemma":[0.99829525,0.0010818362,0.00017575214,0.00018136742,0.00015592233,0.00010989648],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010303849,0.0008060034,0.00059270486,0.00024592315,0.00039585095,0.0006754923,0.0013077329,0.00085139886,0.0038492081],"category_scores_gemma":[0.0043233866,0.00033659543,0.00040520143,0.0002297791,0.0009460435,0.0015462383,0.0011509791,0.0017916323,0.00057220587],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021602555,0.0001634523,0.0008266177,0.00011640897,0.000026740174,0.0001536934,0.00020906766,0.8532411,0.006807208,0.027975328,0.0014208793,0.10884358],"study_design_scores_gemma":[0.000028430988,0.0000402218,0.00003322943,0.00000703686,0.0000046053806,0.00001386336,0.000012527334,0.9883752,0.0019073506,0.008885316,0.0006880828,0.00000409815],"about_ca_topic_score_codex":0.0026075295,"about_ca_topic_score_gemma":0.0037542204,"teacher_disagreement_score":0.0038492081,"about_ca_system_score_codex":0.000849223,"about_ca_system_score_gemma":0.001461535,"threshold_uncertainty_score":0.012876868},"labels":[],"label_agreement":null},{"id":"W2809151193","doi":"10.48550/arxiv.1808.09127","title":"High-confidence error estimates for learned value functions","year":2018,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Bellman equation; Benchmark (surveying); Computer science; Value (mathematics); Convergence (economics); Stability (learning theory); Function (biology); Mathematical optimization; Upper and lower bounds; Artificial intelligence; Machine learning; Mathematics","score_opus":0.08861862077521193,"score_gpt":0.21932372073798673,"score_spread":0.1307050999627748,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2809151193","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00970447,0.0002282708,0.98823416,0.0003151994,0.000035077104,0.00003947505,0.000059265967,0.00053905934,0.00084501295],"genre_scores_gemma":[0.5151451,0.00042294027,0.47980365,0.0003372456,0.0001384061,0.0003747154,0.0006296842,0.000906935,0.002241275],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.989794,0.0035113683,0.0007824513,0.002035278,0.0031722207,0.0007047137],"domain_scores_gemma":[0.83078545,0.13793537,0.006805725,0.012939628,0.010133297,0.0014004704],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015057537,0.0019864729,0.0026406252,0.0017451936,0.0009637493,0.0039546466,0.004771421,0.0034462595,0.0044960487],"category_scores_gemma":[0.20006357,0.0016817253,0.0012702692,0.0011819286,0.0039698686,0.008407442,0.0047423947,0.007989336,0.001396788],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00054265297,0.00024780148,0.0055410676,0.00040772225,0.00016179185,0.00016448852,0.0003189248,0.8073922,0.0041451296,0.070320606,0.0024693697,0.10828822],"study_design_scores_gemma":[0.000024740912,0.000055948334,0.0005270262,0.00007020498,0.00000956362,0.000045958026,0.00002328591,0.96265185,0.003240956,0.03290323,0.00042573662,0.000021529477],"about_ca_topic_score_codex":0.003129006,"about_ca_topic_score_gemma":0.002782106,"teacher_disagreement_score":0.015057537,"about_ca_system_score_codex":0.0031174438,"about_ca_system_score_gemma":0.0027069424,"threshold_uncertainty_score":0.07963282},"labels":[],"label_agreement":null},{"id":"W2809487708","doi":"","title":"Comparing Direct and Indirect Temporal-Difference Methods for Estimating the Variance of the Return","year":2018,"lang":"en","type":"article","venue":"Uncertainty in Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Variance (accounting); Temporal difference learning; Computer science; Reinforcement learning; Mean squared error; Lookup table; Function (biology); Algorithm; Artificial intelligence; Statistics; Mathematics","score_opus":0.10081530290189013,"score_gpt":0.3667202816540406,"score_spread":0.26590497875215047,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2809487708","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028061109,0.00085192773,0.9682406,0.00026714374,0.00007629362,0.00007144466,0.000050787716,0.0004688063,0.0019119154],"genre_scores_gemma":[0.45262772,0.00059823086,0.54157287,0.0002866102,0.00013430389,0.00023455177,0.00033109484,0.00035598068,0.0038586007],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99808013,0.0007154863,0.00012292691,0.00038588574,0.00060041365,0.000095170086],"domain_scores_gemma":[0.9808225,0.015067269,0.000829485,0.001277659,0.0016328646,0.00037031525],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0053409,0.0010836672,0.0011839793,0.00097327674,0.00034562667,0.001517429,0.0020335508,0.001379444,0.0022541583],"category_scores_gemma":[0.03270878,0.00044349107,0.0006960768,0.00066703116,0.00102828,0.0029325737,0.0026426858,0.0024744282,0.0005250929],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007883848,0.00030663746,0.009143627,0.0003544764,0.0003107862,0.000073926414,0.0002900859,0.52133024,0.003762903,0.032735776,0.0021002144,0.42880294],"study_design_scores_gemma":[0.000035017933,0.00007112028,0.0006451336,0.000025094925,0.000020820858,0.000044189714,0.000023441778,0.9880095,0.0015999891,0.008978442,0.0005303762,0.00001681413],"about_ca_topic_score_codex":0.004060122,"about_ca_topic_score_gemma":0.004204431,"teacher_disagreement_score":0.0053409,"about_ca_system_score_codex":0.0010055485,"about_ca_system_score_gemma":0.0017263817,"threshold_uncertainty_score":0.028245747},"labels":[],"label_agreement":null},{"id":"W2817967199","doi":"10.1613/jair.1.12463","title":"The Bottleneck Simulator: A Model-Based Deep Reinforcement Learning Approach","year":2020,"lang":"en","type":"preprint","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Université de Montréal; Polytechnique Montréal; Mila - Quebec Artificial Intelligence Institute","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Compute Canada; Amazon Web Services; Nuance Foundation; Canadian Institute for Advanced Research; Nvidia","keywords":"Bottleneck; Reinforcement learning; Computer science; Variance (accounting); Task (project management); Obstacle; Artificial intelligence; State space; Engineering; Statistics; Mathematics","score_opus":0.20092186190179046,"score_gpt":0.4031124373316161,"score_spread":0.20219057542982566,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2817967199","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019271424,0.00009373725,0.9785452,0.00016791065,0.000029012095,0.000052724747,0.000045002485,0.0009110093,0.00088387355],"genre_scores_gemma":[0.78490114,0.00012531837,0.21186893,0.00020366162,0.00003769122,0.0002822304,0.0001892908,0.00021963897,0.0021720424],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99943346,0.00026708597,0.000022897782,0.00008923807,0.00012347911,0.0000639436],"domain_scores_gemma":[0.9977538,0.0014163131,0.00018484796,0.00021272016,0.00026852035,0.00016380167],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022205336,0.00082570704,0.0011725331,0.0005155936,0.00031285672,0.00067542144,0.0024616227,0.0011944871,0.0020295987],"category_scores_gemma":[0.0054210527,0.0006536679,0.00060761307,0.00032959422,0.00095688127,0.0013800813,0.0016066043,0.002274984,0.00035017292],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000073274496,0.000046230438,0.00039628556,0.000025423506,0.00003113765,0.000024771698,0.000021413802,0.9791047,0.000841835,0.004663364,0.00037797645,0.014393648],"study_design_scores_gemma":[0.000003890369,0.000010502766,0.000010553448,9.832464e-7,0.0000011818338,0.0000016244854,6.2483446e-7,0.9987715,0.00011599105,0.0010397619,0.000042082756,0.0000012229419],"about_ca_topic_score_codex":0.0052162223,"about_ca_topic_score_gemma":0.004409615,"teacher_disagreement_score":0.0052162223,"about_ca_system_score_codex":0.0011686512,"about_ca_system_score_gemma":0.0020602695,"threshold_uncertainty_score":0.011743426},"labels":[],"label_agreement":null},{"id":"W2884439071","doi":"10.1017/s0269888921000035","title":"Safe option-critic: learning safety in the option-critic architecture","year":2021,"lang":"en","type":"article","venue":"The Knowledge Engineering Review","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Reinforcement learning; Consistency (knowledge bases); State space; Variance (accounting); Function (biology); Artificial intelligence; Grid; Architecture; Machine learning; Mathematical optimization; Mathematics","score_opus":0.011610790669513473,"score_gpt":0.25518738082408904,"score_spread":0.24357659015457556,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2884439071","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020876745,0.0013049095,0.971394,0.0007230595,0.000061237006,0.000034586996,0.0000340141,0.00050672144,0.005064765],"genre_scores_gemma":[0.84746504,0.0012317548,0.14584938,0.0002450774,0.000066604225,0.00011894841,0.000072705734,0.00012412066,0.004826389],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994429,0.00028961123,0.000024478944,0.00007544319,0.00012219623,0.0000453819],"domain_scores_gemma":[0.9981077,0.0012551644,0.00014005354,0.00011046768,0.00027106,0.000115598785],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018990347,0.0008115152,0.000737307,0.0004039898,0.0002689522,0.0008881319,0.0015362033,0.0010562354,0.0024309799],"category_scores_gemma":[0.004377479,0.00044143468,0.00044955645,0.00035265792,0.001352541,0.0011774658,0.0007913068,0.0021760634,0.00037442546],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000054393648,0.000027178687,0.00054301985,0.000088523484,0.00005457103,0.00004121206,0.000041970146,0.9240856,0.0006314852,0.038565494,0.00093903125,0.03492761],"study_design_scores_gemma":[0.000012042597,0.000018332059,0.00005728444,0.000013419736,0.000006689853,0.0000093854405,0.0000027033702,0.98021257,0.00027010872,0.018946428,0.00044599723,0.000005000377],"about_ca_topic_score_codex":0.0042300285,"about_ca_topic_score_gemma":0.0034787578,"teacher_disagreement_score":0.0042300285,"about_ca_system_score_codex":0.0011561519,"about_ca_system_score_gemma":0.0011183743,"threshold_uncertainty_score":0.010043144},"labels":[],"label_agreement":null},{"id":"W2885541157","doi":"10.1109/tac.2020.2978037","title":"On Passivity, Reinforcement Learning, and Higher Order Learning in Multiagent Finite Games","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Automatic Control","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Convergence (economics); Monotonic function; Nash equilibrium; Stochastic game; Property (philosophy); Class (philosophy); Exploit","score_opus":0.015711761885379833,"score_gpt":0.24572690650981754,"score_spread":0.2300151446244377,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2885541157","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0069651855,0.00023405148,0.98925656,0.00018539556,0.000023149694,0.000032870044,0.000010907478,0.000046117235,0.0032457325],"genre_scores_gemma":[0.85505867,0.0010367463,0.1379388,0.00023574498,0.00011613467,0.00036492184,0.00004638863,0.000078023426,0.0051245857],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99849415,0.0007262518,0.00007618922,0.00020202417,0.00036679802,0.00013448665],"domain_scores_gemma":[0.99415565,0.0042486875,0.00057193823,0.00035549828,0.0004441917,0.00022403103],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042816526,0.0013563405,0.0012101306,0.0008917022,0.0006105929,0.0014880735,0.0014234905,0.0014155222,0.0023103335],"category_scores_gemma":[0.009662424,0.0005203655,0.0013454189,0.00046322972,0.0041613607,0.0022080778,0.0018059812,0.0026004652,0.00034089165],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000033102348,0.000043385215,0.0003635507,0.00010473476,0.000045721947,0.00009264908,0.00017825449,0.55761784,0.0018097225,0.42896944,0.0002982715,0.010443318],"study_design_scores_gemma":[0.000011318016,0.000065459106,0.00008157533,0.000020675081,0.000007656255,0.000022750422,0.000011905332,0.85297465,0.0005416055,0.14553176,0.0007180693,0.000012671166],"about_ca_topic_score_codex":0.0027820386,"about_ca_topic_score_gemma":0.0014949092,"teacher_disagreement_score":0.0042816526,"about_ca_system_score_codex":0.0020556129,"about_ca_system_score_gemma":0.0015618013,"threshold_uncertainty_score":0.022643805},"labels":[],"label_agreement":null},{"id":"W2886012730","doi":"10.1609/aaai.v34i04.5955","title":"Count-Based Exploration with the Successor Representation","year":2020,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Alberta Innovates; University of Alberta; Alberta Machine Intelligence Institute; Compute Canada","keywords":"Successor cardinal; Representation (politics); Computer science; Norm (philosophy); Generalization; Sample complexity; Reinforcement learning; Similarity (geometry); Artificial intelligence; State (computer science); Algorithm; Theoretical computer science; Mathematics","score_opus":0.15508351119977026,"score_gpt":0.3234339624244677,"score_spread":0.16835045122469744,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2886012730","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029503418,0.00021341247,0.9652595,0.00039585706,0.000057056903,0.00007327984,0.00010708408,0.000958699,0.0034317106],"genre_scores_gemma":[0.67277193,0.00015484623,0.3206093,0.00023335103,0.00006268047,0.00032846257,0.0002834026,0.00021344094,0.0053426055],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989448,0.00038168568,0.00006751131,0.00023256014,0.00026394267,0.000109548666],"domain_scores_gemma":[0.99654764,0.0020721399,0.00031572505,0.00056903233,0.00028464303,0.0002107694],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018416712,0.0007172966,0.0012407788,0.00079417735,0.00050201145,0.0013311476,0.0022965425,0.0012871362,0.0047647236],"category_scores_gemma":[0.008967494,0.00039721426,0.0006705508,0.0008105403,0.0014559806,0.003987537,0.002790785,0.0020429553,0.0006649088],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042185446,0.00017986416,0.0020058544,0.00020080151,0.000067556146,0.00012614268,0.00024039879,0.52987224,0.003811494,0.23789021,0.0035037545,0.22167985],"study_design_scores_gemma":[0.000027210159,0.000068749694,0.0000775646,0.0000132735095,0.000007869891,0.000028893222,0.000009882,0.9336309,0.000837933,0.064494535,0.00079269614,0.000010499109],"about_ca_topic_score_codex":0.0012526028,"about_ca_topic_score_gemma":0.001776898,"teacher_disagreement_score":0.0047647236,"about_ca_system_score_codex":0.0010744798,"about_ca_system_score_gemma":0.001758761,"threshold_uncertainty_score":0.015939653},"labels":[],"label_agreement":null},{"id":"W2889310374","doi":"10.1109/ccece.2018.8447854","title":"Socially Aware Robot Navigation Using Deep Reinforcement Learning","year":2018,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Prince Edward Island","funders":"","keywords":"Reinforcement learning; Mobile robot; Computer science; Social robot; Robot; Artificial intelligence; Mobile robot navigation; Robot learning; Human–computer interaction; Human–robot interaction; Robot control","score_opus":0.028587905249804244,"score_gpt":0.2871830959177124,"score_spread":0.25859519066790815,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2889310374","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04088325,0.00019267765,0.9558735,0.00025422656,0.00005039765,0.000029741197,0.00001766669,0.000339653,0.0023589262],"genre_scores_gemma":[0.94715214,0.00009818708,0.050391395,0.00011641094,0.000021085218,0.000057168716,0.000033451073,0.000025097936,0.0021051115],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99974126,0.00008440972,0.0000096189415,0.00006158908,0.000056798184,0.000046186375],"domain_scores_gemma":[0.9995783,0.00016504478,0.00007411421,0.000034833098,0.00010068358,0.000047069218],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00050235493,0.0006758548,0.00068322493,0.0001955156,0.0003237083,0.0004548864,0.000959575,0.00077607145,0.000989582],"category_scores_gemma":[0.0013165013,0.00024631692,0.00034947775,0.00013364691,0.00080679177,0.0006431466,0.0008470922,0.0008637263,0.00018062342],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000049967453,0.0000620089,0.0010592036,0.000034651006,0.00003461428,0.000077561854,0.000067796806,0.95784444,0.002485348,0.008368098,0.00052427035,0.029392183],"study_design_scores_gemma":[0.0000042010047,0.000018593933,0.00004842367,0.0000018974893,0.0000029788193,0.0000066471043,0.000004393824,0.99746966,0.00021498687,0.0020609403,0.00016501828,0.0000022607696],"about_ca_topic_score_codex":0.0056428127,"about_ca_topic_score_gemma":0.005080191,"teacher_disagreement_score":0.0056428127,"about_ca_system_score_codex":0.00075919944,"about_ca_system_score_gemma":0.0009774938,"threshold_uncertainty_score":0.011219919},"labels":[],"label_agreement":null},{"id":"W2889665471","doi":"10.1109/agents.2018.8460004","title":"Autonomous Agents in Snake Game via Deep Reinforcement Learning","year":2018,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"BC Research (Canada)","funders":"People's Government of Jilin Province; National Research Foundation","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Constraint (computer-aided design); Dual (grammatical number); Machine learning; Engineering","score_opus":0.024876329767722426,"score_gpt":0.2714467120606008,"score_spread":0.24657038229287837,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2889665471","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15959643,0.00033769986,0.83154273,0.00043909135,0.00008054377,0.00010849745,0.000043027852,0.0009179963,0.006933987],"genre_scores_gemma":[0.96894014,0.00006286346,0.028745878,0.00008203657,0.000008137157,0.00006822261,0.000023383474,0.00002028751,0.0020490433],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99983704,0.00004370501,0.000007968236,0.00003602919,0.00003588609,0.000039291903],"domain_scores_gemma":[0.99961406,0.00017605044,0.00006403229,0.000027363023,0.00006214336,0.000056456534],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005488487,0.00069402927,0.00063560915,0.00021330427,0.00026837815,0.00047428475,0.00085321604,0.0006096628,0.0015159011],"category_scores_gemma":[0.0015272247,0.00032061661,0.00030663106,0.00011954909,0.0006312895,0.0006604081,0.00091147015,0.0010641665,0.00016819153],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006683307,0.00006470879,0.0009112631,0.000028151662,0.000019768368,0.000070385766,0.000049837545,0.9680113,0.0018079241,0.0052492805,0.00046174356,0.0232587],"study_design_scores_gemma":[0.0000050993376,0.000018564313,0.00003985689,0.0000018076902,0.0000019121185,0.00000445315,0.0000026610023,0.99845684,0.00017643254,0.0011925763,0.0000981893,0.0000015951989],"about_ca_topic_score_codex":0.0060513094,"about_ca_topic_score_gemma":0.0049107927,"teacher_disagreement_score":0.0060513094,"about_ca_system_score_codex":0.00061593077,"about_ca_system_score_gemma":0.00081337395,"threshold_uncertainty_score":0.012032151},"labels":[],"label_agreement":null},{"id":"W2889732123","doi":"10.1609/aaai.v33i01.33013582","title":"Combined Reinforcement Learning via Abstract Representations","year":2019,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Modularity (biology); Bridging (networking); Generalization; Artificial intelligence; Representation (politics); Encoding (memory); Machine learning; Theoretical computer science; Mathematics","score_opus":0.08144392931513404,"score_gpt":0.31546580598770657,"score_spread":0.23402187667257252,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2889732123","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009697022,0.00014058653,0.98812604,0.00018864802,0.000028862869,0.000022068387,0.000044335582,0.00040859182,0.0013439471],"genre_scores_gemma":[0.82628393,0.00024841464,0.16985108,0.00015809723,0.000057872516,0.00018417368,0.00019397671,0.00009997062,0.002922389],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99918205,0.00031088453,0.00004450377,0.00018671268,0.00020179313,0.00007402126],"domain_scores_gemma":[0.9984164,0.00081658066,0.00016191187,0.00033828564,0.00016944011,0.000097400334],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011156969,0.0008902925,0.0011428789,0.00048436841,0.00026938334,0.001152195,0.0015210008,0.0010161811,0.002411722],"category_scores_gemma":[0.004874594,0.00043283918,0.0006690721,0.0005017414,0.0012996707,0.0026032731,0.00244879,0.002205943,0.00043566793],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011315918,0.000066474015,0.00044006816,0.00009434177,0.00006832677,0.0000690148,0.00008418022,0.8272984,0.002515244,0.08750541,0.0010231409,0.08072221],"study_design_scores_gemma":[0.000011417676,0.000029185298,0.000040947703,0.0000055272494,0.000006683923,0.000008489867,0.000003897196,0.9398499,0.00041146114,0.059192095,0.0004343658,0.0000060065736],"about_ca_topic_score_codex":0.0017913347,"about_ca_topic_score_gemma":0.0019453733,"teacher_disagreement_score":0.002411722,"about_ca_system_score_codex":0.00094022957,"about_ca_system_score_gemma":0.0007981521,"threshold_uncertainty_score":0.008068025},"labels":[],"label_agreement":null},{"id":"W2890043461","doi":"10.1609/aiide.v14i1.13030","title":"Improbotics: Exploring the Imitation Game Using Machine Intelligence in Improvised Theatre","year":2018,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Imitation; Turing test; Computer science; Improvisation; Test (biology); Human–computer interaction; Narrative; Perception; Context (archaeology); Control (management); Entertainment; Artificial intelligence; Multimedia; Psychology; Visual arts; Art; Social psychology","score_opus":0.07447162716838729,"score_gpt":0.29601916611341855,"score_spread":0.22154753894503126,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2890043461","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.51159596,0.0009102834,0.455248,0.00093040225,0.00009430719,0.00069588213,0.00015815684,0.0014705549,0.02889643],"genre_scores_gemma":[0.833651,0.00020748569,0.16195174,0.0001620431,0.00001917459,0.0003051333,0.00011409313,0.0001267134,0.0034626606],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99865544,0.00093697803,0.00003295091,0.00013830513,0.0001261469,0.000110053195],"domain_scores_gemma":[0.9983203,0.0013438459,0.00006709688,0.00009188487,0.00005701084,0.00011987543],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014517723,0.00078757555,0.00041353385,0.00037796533,0.0005079122,0.0020435008,0.0016342563,0.0013073793,0.0032867908],"category_scores_gemma":[0.0040622167,0.0004019419,0.00066074054,0.00015435321,0.0021985874,0.0020941712,0.002024438,0.0009790793,0.0004950556],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003234088,0.0032818345,0.015798045,0.0028011557,0.00050102954,0.0036183782,0.03437987,0.4230644,0.14830858,0.11784696,0.0064577963,0.24070795],"study_design_scores_gemma":[0.00021911973,0.0014188528,0.0044893846,0.0001558114,0.000065041924,0.00069887657,0.0036015934,0.9159735,0.0140286265,0.038448315,0.020785058,0.000115857256],"about_ca_topic_score_codex":0.0018984464,"about_ca_topic_score_gemma":0.0027639936,"teacher_disagreement_score":0.0032867908,"about_ca_system_score_codex":0.0006897101,"about_ca_system_score_gemma":0.00054896175,"threshold_uncertainty_score":0.010995388},"labels":[],"label_agreement":null},{"id":"W2890047470","doi":"10.48550/arxiv.1811.09013","title":"An Off-policy Policy Gradient Theorem Using Emphatic Weightings","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Unicode; Counterexample; Reinforcement learning; Computer science; Key (lock); Mathematical economics; Mathematics; Mathematical optimization; Artificial intelligence; Discrete mathematics; Computer security","score_opus":0.06837137357235425,"score_gpt":0.23102146841004043,"score_spread":0.16265009483768617,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2890047470","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005860225,0.00008148948,0.9894386,0.0002858247,0.00004882177,0.000031192667,0.000018210232,0.00015205274,0.004083597],"genre_scores_gemma":[0.5574764,0.00037144695,0.42797926,0.00065697636,0.00013595162,0.00027937765,0.00011767427,0.00037588028,0.01260697],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99913776,0.00033050985,0.000039820796,0.00016397929,0.0002519081,0.00007603981],"domain_scores_gemma":[0.99789786,0.0012772669,0.00017603957,0.00024089102,0.0003058699,0.00010202195],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027047046,0.0010122181,0.00089996157,0.0004352611,0.00043035764,0.0011079663,0.0012534442,0.0013449346,0.0045328727],"category_scores_gemma":[0.01148087,0.00057293446,0.00053214486,0.00032555932,0.0020465867,0.0022434935,0.00193892,0.0026686494,0.000783656],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012262493,0.00006136678,0.0007815849,0.0001379147,0.000045020293,0.00011771308,0.000154199,0.49767867,0.0039441413,0.4278703,0.0035653524,0.065521106],"study_design_scores_gemma":[0.000015488884,0.000039659146,0.00008173227,0.000015557218,0.0000056764015,0.000031151656,0.0000068208738,0.9384211,0.00085312163,0.059062943,0.0014588303,0.000007990525],"about_ca_topic_score_codex":0.0016995958,"about_ca_topic_score_gemma":0.0014838568,"teacher_disagreement_score":0.0045328727,"about_ca_system_score_codex":0.000983371,"about_ca_system_score_gemma":0.0014439793,"threshold_uncertainty_score":0.015163958},"labels":[],"label_agreement":null},{"id":"W2890185089","doi":"","title":"Monte-Carlo Tree Search for Constrained POMDPs","year":2018,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Monte Carlo tree search; Computer science; Partially observable Markov decision process; Mathematical optimization; Monte Carlo method; Markov decision process; Tree (set theory); Action selection; Scale (ratio); Markov process; Machine learning; Markov model; Mathematics; Markov chain","score_opus":0.035799904152969325,"score_gpt":0.284999641127341,"score_spread":0.24919973697437167,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2890185089","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014603129,0.00036126876,0.9796382,0.0002644672,0.00004790552,0.00010627401,0.00012543653,0.00037391923,0.00447935],"genre_scores_gemma":[0.60248697,0.00041923823,0.3921625,0.00029476805,0.00006989703,0.0007206461,0.00047979565,0.00025156248,0.003114547],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99920505,0.00033069955,0.000042390497,0.00013940701,0.0001802721,0.00010226252],"domain_scores_gemma":[0.9951891,0.004044799,0.00022385365,0.00012550359,0.00025729917,0.00015945414],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017450101,0.00125677,0.0016789745,0.0007232064,0.00070134277,0.0010217481,0.0014245537,0.0016207634,0.0057218797],"category_scores_gemma":[0.007960003,0.0008343482,0.0009789519,0.0008469151,0.001343438,0.00141482,0.0015558018,0.0023142924,0.00050336006],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003403025,0.000020474336,0.00022914966,0.000050341918,0.000015622025,0.000031812415,0.000021554299,0.97831446,0.0001450426,0.0133619895,0.00048789385,0.0072876657],"study_design_scores_gemma":[0.0000100963725,0.000006904401,0.0000170352,0.000004704245,0.0000024497901,0.0000037041452,0.0000028112331,0.99379736,0.00004720106,0.0059033753,0.00020257826,0.0000017095425],"about_ca_topic_score_codex":0.0084951855,"about_ca_topic_score_gemma":0.009017726,"teacher_disagreement_score":0.0084951855,"about_ca_system_score_codex":0.0015083904,"about_ca_system_score_gemma":0.0026454306,"threshold_uncertainty_score":0.019141555},"labels":[],"label_agreement":null},{"id":"W2890960027","doi":"","title":"Iterative Value-Aware Model Learning","year":2018,"lang":"en","type":"article","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute","funders":"","keywords":"Reinforcement learning; Computer science; Contrast (vision); Class (philosophy); Mathematical optimization; Artificial intelligence; Value (mathematics); Iterative learning control; Iterative method; Optimization problem; Decision problem; Machine learning; Algorithm; Mathematics; Control (management)","score_opus":0.014372635973734319,"score_gpt":0.24675203116280983,"score_spread":0.23237939518907552,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2890960027","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005705336,0.00011158628,0.991682,0.00015982472,0.00001460545,0.000029068657,0.000019114246,0.00030177354,0.0019767357],"genre_scores_gemma":[0.70907307,0.0002283454,0.28597873,0.00026197123,0.000049667815,0.00027200862,0.0001865486,0.00018535885,0.0037642817],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983145,0.0006559777,0.00006335948,0.00032100466,0.0004343029,0.0002109561],"domain_scores_gemma":[0.9948791,0.0036815389,0.0003372892,0.0005158139,0.0004220591,0.00016413578],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020619435,0.0011916981,0.0015488325,0.0005377662,0.0005091616,0.0013728236,0.002272788,0.0015359451,0.0029872097],"category_scores_gemma":[0.010293167,0.0007849041,0.000815039,0.00056581903,0.0015582942,0.00218849,0.0030054215,0.002664538,0.0005542154],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008489956,0.00007035461,0.00061103905,0.00009448225,0.000045394776,0.000077128214,0.00011999562,0.87802,0.00095896685,0.059859112,0.0015581561,0.058500502],"study_design_scores_gemma":[0.000007209645,0.00001726959,0.000023456758,0.0000058579462,0.000003542204,0.000012354844,0.000004904847,0.9806629,0.00027229748,0.018646844,0.0003400855,0.0000032232826],"about_ca_topic_score_codex":0.0032930905,"about_ca_topic_score_gemma":0.0033101833,"teacher_disagreement_score":0.0032930905,"about_ca_system_score_codex":0.0014310906,"about_ca_system_score_gemma":0.002188389,"threshold_uncertainty_score":0.010904729},"labels":[],"label_agreement":null},{"id":"W2891166116","doi":"10.48550/arxiv.1811.00429","title":"Temporal Regularization in Markov Decision Process","year":2018,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval; McGill University","funders":"","keywords":"Regularization (linguistics); Markov decision process; Reinforcement learning; Computer science; Exploit; Artificial intelligence; Machine learning; Markov process; Mathematical optimization; Partially observable Markov decision process; Markov chain; Regularization perspectives on support vector machines; Inverse problem; Mathematics; Markov model; Statistics","score_opus":0.03660652369230544,"score_gpt":0.1961727198969281,"score_spread":0.15956619620462267,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2891166116","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018624749,0.0011853565,0.9723397,0.0013785401,0.000102720674,0.00005382689,0.00010278182,0.00016820607,0.0060440907],"genre_scores_gemma":[0.88980764,0.0015292218,0.099981084,0.0004568769,0.00022042086,0.0003201824,0.00017316468,0.00008399674,0.007427487],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977496,0.0011772445,0.00009689582,0.00038375158,0.00041609464,0.0001763705],"domain_scores_gemma":[0.98803097,0.010155823,0.0007266156,0.0003164727,0.00052155193,0.0002486075],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004554663,0.00095956173,0.0014849756,0.0006427149,0.00070897647,0.0013482845,0.0011918077,0.0019012304,0.00407268],"category_scores_gemma":[0.014636492,0.00058604946,0.0008684356,0.0008056529,0.0021185009,0.0018843514,0.0015328781,0.0028950104,0.00036275308],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000824286,0.000054715096,0.0010376583,0.0001434049,0.000058496695,0.00012796583,0.0001317674,0.6060611,0.00064162567,0.37267423,0.0013602078,0.017626375],"study_design_scores_gemma":[0.000012907937,0.00001978165,0.000097349475,0.0000124547205,0.0000058761616,0.000010596415,0.0000057177167,0.89165413,0.00010102317,0.10761583,0.0004567654,0.0000075757703],"about_ca_topic_score_codex":0.0077605946,"about_ca_topic_score_gemma":0.0056702523,"teacher_disagreement_score":0.0077605946,"about_ca_system_score_codex":0.0023942466,"about_ca_system_score_gemma":0.0017223506,"threshold_uncertainty_score":0.024087608},"labels":[],"label_agreement":null},{"id":"W2891236810","doi":"","title":"Reinforcement Learning with Multiple Experts: A Bayesian Model Combination Approach","year":2018,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Convergence (economics); Artificial intelligence; Machine learning; Bayesian probability; Bellman equation; Domain (mathematical analysis); Function (biology); Mathematical optimization; Mathematics","score_opus":0.019661178052500883,"score_gpt":0.23828292618312835,"score_spread":0.21862174813062746,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2891236810","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0070281164,0.0002055434,0.9899048,0.00030138792,0.000026845686,0.000047290763,0.000016719783,0.00015220161,0.002317084],"genre_scores_gemma":[0.74878913,0.0003916981,0.24445309,0.00042628066,0.00015696854,0.00039860545,0.000082676546,0.000104916,0.0051965774],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99713516,0.0014590321,0.00009654789,0.0003901598,0.00069502153,0.00022406522],"domain_scores_gemma":[0.9952195,0.0031629242,0.0005304661,0.0003020804,0.0005295891,0.00025535235],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0048101726,0.0016041313,0.002525439,0.0010593882,0.00061573996,0.0014573871,0.002973491,0.0026477997,0.0032766825],"category_scores_gemma":[0.010635639,0.0011777972,0.0011756765,0.0007915839,0.0017450949,0.0025087746,0.0025450967,0.0032384743,0.00068470556],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007446039,0.00005781961,0.00038475028,0.000051097064,0.00009488788,0.00007553061,0.000072840056,0.9492175,0.00064597395,0.023350416,0.0006464614,0.025328344],"study_design_scores_gemma":[0.000012723538,0.00002440352,0.000040725692,0.000007131869,0.000011179395,0.000015024368,0.0000035892085,0.9881025,0.00015290751,0.011357727,0.00026369328,0.000008388519],"about_ca_topic_score_codex":0.0026028585,"about_ca_topic_score_gemma":0.0028805523,"teacher_disagreement_score":0.0048101726,"about_ca_system_score_codex":0.0013871018,"about_ca_system_score_gemma":0.0014419059,"threshold_uncertainty_score":0.025438905},"labels":[],"label_agreement":null},{"id":"W2895478303","doi":"10.48550/arxiv.1810.01032","title":"Reinforcement Learning with Perturbed Rewards","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Noise (video); Confusion matrix; Artificial intelligence; Convergence (economics); Confusion; Set (abstract data type); Gaussian; Machine learning; Matrix (chemical analysis); Mathematical optimization; Algorithm; Mathematics","score_opus":0.053332976380586775,"score_gpt":0.18502390545204497,"score_spread":0.13169092907145818,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2895478303","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027472874,0.00034255875,0.96911174,0.00038177834,0.00005474201,0.00007188218,0.00006345378,0.0005808074,0.0019201499],"genre_scores_gemma":[0.92217773,0.00017457482,0.074591346,0.00025263027,0.000056361318,0.00017446198,0.00013482045,0.00008646631,0.0023515515],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99771035,0.0009557036,0.00012427602,0.00053634634,0.00043146685,0.00024178447],"domain_scores_gemma":[0.9935306,0.0041874554,0.000771853,0.00048023034,0.00074170146,0.00028819081],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027958532,0.0014273953,0.001592421,0.000480973,0.00044246874,0.0011851405,0.0014421222,0.0013007511,0.0019236057],"category_scores_gemma":[0.014687581,0.0006194672,0.00053645,0.00037032992,0.0019534058,0.0014908953,0.0016007829,0.0022493375,0.00041495386],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014053493,0.00005542979,0.00095914636,0.00007013173,0.00004395815,0.0000690605,0.00006553119,0.9587994,0.0008603268,0.015456766,0.0007183386,0.022761364],"study_design_scores_gemma":[0.000013170348,0.000024781173,0.00007962972,0.0000065424797,0.000005109887,0.0000097187285,0.0000040916807,0.99177575,0.0002916467,0.0075798654,0.00020489209,0.000004630415],"about_ca_topic_score_codex":0.004046855,"about_ca_topic_score_gemma":0.0029070585,"teacher_disagreement_score":0.004046855,"about_ca_system_score_codex":0.001549922,"about_ca_system_score_gemma":0.0016455889,"threshold_uncertainty_score":0.014786124},"labels":[],"label_agreement":null},{"id":"W2898050260","doi":"10.1109/ijcnn.2018.8489122","title":"Mixing Habits and Planning for Multi-Step Target Reaching Using Arbitrated Predictive Actor-Critic","year":2018,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Task (project management); Control (management); Model predictive control; Internal model; Controller (irrigation); Artificial intelligence; Work (physics); Engineering","score_opus":0.07409115597081546,"score_gpt":0.32788170110033965,"score_spread":0.2537905451295242,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2898050260","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.29975888,0.00025702882,0.695957,0.00027455733,0.00004285205,0.00009693766,0.000032385808,0.0006608088,0.0029195247],"genre_scores_gemma":[0.97764707,0.000039427603,0.021739146,0.000016727447,0.0000036477388,0.000048331673,0.0000148267945,0.000013552463,0.00047726813],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99973434,0.00008792813,0.000017869661,0.00007256377,0.000055217883,0.00003212281],"domain_scores_gemma":[0.9989812,0.0005852581,0.00015136515,0.00011221256,0.00010212793,0.000067768124],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00097579614,0.00053156464,0.00037362924,0.00028189644,0.00026925496,0.00054083654,0.0007602887,0.0005938666,0.0008795874],"category_scores_gemma":[0.0030039432,0.00046423933,0.000364404,0.00021240034,0.0007587615,0.0007653098,0.000604899,0.0010855742,0.00010234746],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013631953,0.00007503645,0.0021661948,0.00005603118,0.00006983641,0.000090350906,0.00012705821,0.96613073,0.005078782,0.003760396,0.00016503384,0.022144264],"study_design_scores_gemma":[0.000005915713,0.000028770524,0.000241838,0.0000015432558,0.000005809226,0.000007630158,0.000003045505,0.99791604,0.00054254994,0.0011880054,0.00005534115,0.0000035086157],"about_ca_topic_score_codex":0.0052256254,"about_ca_topic_score_gemma":0.005102076,"teacher_disagreement_score":0.0052256254,"about_ca_system_score_codex":0.00067201554,"about_ca_system_score_gemma":0.0007037843,"threshold_uncertainty_score":0.010390401},"labels":[],"label_agreement":null},{"id":"W2900798285","doi":"10.65109/loxq3741","title":"Urban Driving with Multi-Objective Deep Reinforcement Learning","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Task (project management); Artificial intelligence; Domain (mathematical analysis); Dual (grammatical number); Markov decision process; Function (biology); Lexicographical order; Collision; Markov process; Machine learning; Computer security; Engineering; Mathematics","score_opus":0.016748048215011115,"score_gpt":0.24694989287972507,"score_spread":0.23020184466471397,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2900798285","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07895666,0.00025733892,0.91559666,0.00048133076,0.00006544072,0.000056030545,0.00007167653,0.00081081974,0.003703985],"genre_scores_gemma":[0.94593775,0.000050726732,0.05147228,0.00011828168,0.000018429739,0.000050351464,0.00007795732,0.00003223545,0.0022420823],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997931,0.00006556689,0.0000091482425,0.00004996188,0.00004254484,0.00003968597],"domain_scores_gemma":[0.99943,0.00029697252,0.000058680314,0.000057899844,0.00010895189,0.000047450212],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000846913,0.00052079145,0.0005106929,0.00017391387,0.00020264488,0.00044216725,0.0008898489,0.00073920126,0.0018176723],"category_scores_gemma":[0.0020588553,0.00033502965,0.0002597933,0.00021153493,0.0006944421,0.0006999109,0.0009305014,0.0012672661,0.00023880636],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004510696,0.00005323563,0.0005359584,0.0000215233,0.00001812077,0.0000276944,0.000017508968,0.9707963,0.0007668988,0.0037419263,0.00052958645,0.023446182],"study_design_scores_gemma":[0.0000037697482,0.000008753976,0.000023490704,8.1114024e-7,8.0570334e-7,0.0000017375878,9.824967e-7,0.9985794,0.000106181484,0.0012140266,0.000059172322,8.2711017e-7],"about_ca_topic_score_codex":0.006512761,"about_ca_topic_score_gemma":0.0066356948,"teacher_disagreement_score":0.006512761,"about_ca_system_score_codex":0.00089615176,"about_ca_system_score_gemma":0.0009950281,"threshold_uncertainty_score":0.012949705},"labels":[],"label_agreement":null},{"id":"W2900994167","doi":"","title":"A Fitted-Q Algorithm for Budgeted MDPs","year":2018,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Algorithm; Mathematical optimization; Mathematics","score_opus":0.016624235311717504,"score_gpt":0.24256595073113119,"score_spread":0.22594171541941369,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2900994167","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0070691933,0.000279191,0.98902136,0.0003450634,0.00009762215,0.00012888742,0.00009170577,0.0004189962,0.0025479717],"genre_scores_gemma":[0.31760797,0.00028559892,0.67316884,0.00039732887,0.0001378802,0.0008033079,0.0004425073,0.00036437466,0.006792179],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99867237,0.0005747839,0.000080679114,0.0002575053,0.00024443623,0.00017020127],"domain_scores_gemma":[0.99339384,0.0049499283,0.00024789467,0.0002941058,0.0007179651,0.0003962812],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037824176,0.0015081003,0.0028582148,0.0012165942,0.0008701415,0.0016878607,0.0033168558,0.004157971,0.012178267],"category_scores_gemma":[0.015363481,0.0015675282,0.0011702663,0.0013568256,0.0017857024,0.0018921396,0.003325962,0.0033691048,0.0015363824],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001180589,0.00006258651,0.00023311285,0.00008761036,0.00003241073,0.000046212917,0.00004056824,0.957713,0.00022808749,0.010661525,0.0014075177,0.02936939],"study_design_scores_gemma":[0.000034541757,0.00001988153,0.000020063295,0.000009237283,0.00000395767,0.0000059005674,0.0000045003426,0.9948791,0.000053901593,0.0047160243,0.00024930382,0.000003550323],"about_ca_topic_score_codex":0.010897603,"about_ca_topic_score_gemma":0.008188355,"teacher_disagreement_score":0.012178267,"about_ca_system_score_codex":0.0019223995,"about_ca_system_score_gemma":0.0036837817,"threshold_uncertainty_score":0.04074037},"labels":[],"label_agreement":null},{"id":"W2901559512","doi":"10.22215/etd/2013-09907","title":"Study of Multiple Multiagent Reinforcement Learning Algorithms in Grid Games","year":2013,"lang":"en","type":"dissertation","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Reinforcement learning; Nash equilibrium; Minimax; Computer science; Algorithm; Best response; Grid; Game theory; Mathematical optimization; Artificial intelligence; Mathematics; Mathematical economics","score_opus":0.024807483136685807,"score_gpt":0.28220742561284134,"score_spread":0.25739994247615555,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2901559512","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10470107,0.0013849026,0.8743282,0.0013569782,0.00013474868,0.00014082727,0.000037326474,0.00012828951,0.017787594],"genre_scores_gemma":[0.9292274,0.00064767624,0.06477211,0.000110105844,0.000075410746,0.00018394525,0.00003417732,0.000051210467,0.004897865],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99917954,0.00048566045,0.000028157328,0.0000936673,0.0001356412,0.00007730232],"domain_scores_gemma":[0.9945602,0.0045161787,0.0002893166,0.0001315333,0.0002840716,0.00021877869],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019337723,0.0006432664,0.001063213,0.00043291677,0.00047916564,0.0011981769,0.0012042497,0.0008433882,0.00235038],"category_scores_gemma":[0.009643539,0.0003523657,0.0005307882,0.00048170504,0.0011967157,0.0015350905,0.00090319675,0.0012715044,0.00018337778],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000047471203,0.000053640135,0.00067472475,0.00007226045,0.000045521407,0.000051866245,0.000075591495,0.9039366,0.00028295157,0.08445425,0.00056643586,0.009738658],"study_design_scores_gemma":[0.0000122634665,0.0000181611,0.00005209183,0.0000068418162,0.0000029330297,0.0000063698894,0.000010398989,0.98261744,0.000065229346,0.016933104,0.00027299012,0.0000022038805],"about_ca_topic_score_codex":0.0036838972,"about_ca_topic_score_gemma":0.0015508116,"teacher_disagreement_score":0.0036838972,"about_ca_system_score_codex":0.0011450938,"about_ca_system_score_gemma":0.0010812439,"threshold_uncertainty_score":0.010226846},"labels":[],"label_agreement":null},{"id":"W2902062907","doi":"","title":"Nonlinear Optimization and Symbolic Dynamic Programming for Parameterized Hybrid Markov Decision Processes.","year":2017,"lang":"en","type":"article","venue":"National Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Parameterized complexity; Markov decision process; Computer science; Mathematical optimization; Dynamic programming; Nonlinear system; Markov process; Markov chain; Nonlinear programming; Mathematics; Algorithm; Machine learning","score_opus":0.08557820677108352,"score_gpt":0.3627951243467675,"score_spread":0.27721691757568395,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2902062907","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028529838,0.0015415794,0.9539257,0.0012874895,0.00009917554,0.000052722186,0.00024240644,0.00018708188,0.014133949],"genre_scores_gemma":[0.9059964,0.00083802023,0.080821946,0.00017626303,0.00007503984,0.00025388313,0.00033075933,0.000110021545,0.011397591],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99948275,0.00027507576,0.00002356709,0.00006779472,0.00007760008,0.000073179726],"domain_scores_gemma":[0.994964,0.004326458,0.00030696712,0.00009223806,0.00017636668,0.00013394524],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013445498,0.000988369,0.0013146062,0.00054217695,0.0005372277,0.0012918352,0.0009845031,0.0012926698,0.0054010185],"category_scores_gemma":[0.007868115,0.0006606678,0.00086676935,0.00063293596,0.0016428885,0.0015352381,0.0017109222,0.0020298248,0.0004076897],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005082635,0.000028037648,0.00032768035,0.000056670513,0.00004454075,0.00005034754,0.000039351668,0.8913909,0.00018666638,0.10076476,0.0009010046,0.006159245],"study_design_scores_gemma":[0.0000051166408,0.0000043084383,0.000028109742,0.000004082132,0.0000031401103,0.0000023322148,0.0000047150284,0.96736664,0.000028360892,0.03241411,0.00013636447,0.0000026215716],"about_ca_topic_score_codex":0.0149134165,"about_ca_topic_score_gemma":0.014791153,"teacher_disagreement_score":0.0149134165,"about_ca_system_score_codex":0.002047401,"about_ca_system_score_gemma":0.0021584826,"threshold_uncertainty_score":0.029653192},"labels":[],"label_agreement":null},{"id":"W2902729397","doi":"10.1613/jair.1.11263","title":"Grounding Language for Transfer in Deep Reinforcement Learning","year":2018,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Regina","funders":"","keywords":"Reinforcement learning; Computer science; Variety (cybernetics); Transfer of learning; Representation (politics); Artificial intelligence; Task (project management); Differentiable function; Domain (mathematical analysis); Component (thermodynamics); Meaning (existential); Machine learning","score_opus":0.15032975293862594,"score_gpt":0.4297375410971567,"score_spread":0.27940778815853073,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2902729397","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01679389,0.00012598523,0.97966504,0.00036858118,0.0000445777,0.000046462927,0.000053484568,0.0007189062,0.0021831589],"genre_scores_gemma":[0.88727224,0.00013462562,0.108654164,0.0002729237,0.000039132243,0.00024876758,0.00013108856,0.00014372896,0.0031033184],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99937683,0.00027128306,0.00003212376,0.00015730904,0.000105930856,0.000056572742],"domain_scores_gemma":[0.99786633,0.0014664843,0.00019536001,0.00025755673,0.00014568826,0.00006858312],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015387996,0.00082286296,0.0006731302,0.0003128105,0.00031874835,0.0009020603,0.0014552514,0.0009878164,0.0038699256],"category_scores_gemma":[0.008419376,0.00037676233,0.00046458046,0.00032371905,0.0016158243,0.002792707,0.0016389193,0.0022858505,0.0006117422],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010574236,0.00011970233,0.00055122864,0.0001001776,0.00002803802,0.0001357105,0.00022033199,0.859905,0.00339498,0.077627555,0.0012254252,0.05658612],"study_design_scores_gemma":[0.000009332321,0.000029290677,0.000028814573,0.000007184947,0.0000033413874,0.000008356428,0.000008064091,0.9723651,0.00069735834,0.026296433,0.0005424908,0.0000042166407],"about_ca_topic_score_codex":0.0021774222,"about_ca_topic_score_gemma":0.0020205623,"teacher_disagreement_score":0.0038699256,"about_ca_system_score_codex":0.0011664757,"about_ca_system_score_gemma":0.0010010636,"threshold_uncertainty_score":0.012946188},"labels":[],"label_agreement":null},{"id":"W2903017298","doi":"10.1609/aaai.v33i01.33014512","title":"State-Augmentation Transformations for Risk-Sensitive Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"China Scholarship Council","keywords":"Reinforcement learning; Markov decision process; Markov chain; Successor cardinal; Markov process; Function (biology); Computer science; Bellman equation; Q-learning; State (computer science); Mathematical optimization; Mathematics; Artificial intelligence; Machine learning; Algorithm; Statistics","score_opus":0.043402280773973005,"score_gpt":0.2852395603095862,"score_spread":0.24183727953561318,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2903017298","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005535051,0.00014883316,0.9915959,0.0001459063,0.000030252257,0.000045644014,0.000036827616,0.00018493462,0.0022765459],"genre_scores_gemma":[0.7362184,0.00046738164,0.25702554,0.00027206636,0.00009161309,0.0005958988,0.00022469486,0.00017664781,0.004927811],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99824834,0.000883098,0.00010703881,0.0003266089,0.0003315052,0.000103456376],"domain_scores_gemma":[0.99706537,0.0020458857,0.00023714975,0.0002861625,0.00026355364,0.000101980695],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027869788,0.001322722,0.0010154045,0.000545535,0.00033456393,0.0010587677,0.0009290361,0.0010978951,0.0036842455],"category_scores_gemma":[0.0074072904,0.00042245555,0.001275203,0.00055600726,0.0020261921,0.0018654747,0.0017646059,0.0028030016,0.0005938086],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000090556925,0.000090259135,0.00042442235,0.00011694053,0.000046606798,0.00012345848,0.00018805287,0.7082975,0.0022953816,0.24524248,0.0010764045,0.042007975],"study_design_scores_gemma":[0.000012918,0.00004737906,0.000047191108,0.000010772034,0.000006294512,0.000017014074,0.0000056829795,0.91033375,0.0004909163,0.08820883,0.0008113382,0.000008042922],"about_ca_topic_score_codex":0.0012488537,"about_ca_topic_score_gemma":0.0007517368,"teacher_disagreement_score":0.0036842455,"about_ca_system_score_codex":0.0011842054,"about_ca_system_score_gemma":0.0012391254,"threshold_uncertainty_score":0.014739156},"labels":[],"label_agreement":null},{"id":"W2903094830","doi":"","title":"First, Scale Up to the Robotic Turing Test, Then Worry about Feeling.","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Feeling; Turing test; Consciousness; Worry; Test (biology); Turing; Psychology; Scale (ratio); Cognitive psychology; Cognition; Power (physics); Computer science; Cognitive science; Social psychology; Artificial intelligence","score_opus":0.01600279103353435,"score_gpt":0.24775324926058973,"score_spread":0.23175045822705537,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2903094830","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023769643,0.011098469,0.15874055,0.65096617,0.007998609,0.0002918663,0.00074769126,0.00117638,0.14521071],"genre_scores_gemma":[0.65318835,0.010040088,0.15115818,0.117727295,0.006400531,0.0017300331,0.0011553155,0.000992422,0.057607803],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99605775,0.001274285,0.00029817547,0.0008513896,0.0012136179,0.00030479155],"domain_scores_gemma":[0.9758748,0.012003932,0.001489372,0.00469392,0.004653642,0.0012842495],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0072134407,0.0013904347,0.0013566634,0.0012493916,0.0026548163,0.0038464537,0.0024583952,0.0052621076,0.02507167],"category_scores_gemma":[0.06745452,0.00039009852,0.0014830973,0.0007469702,0.014524887,0.017922707,0.005220014,0.012396352,0.009377899],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023484511,0.00019704892,0.0028386563,0.00065997493,0.00012788537,0.00031037122,0.0011320842,0.0022927318,0.0012613724,0.83455044,0.08654359,0.06985095],"study_design_scores_gemma":[0.000030011108,0.000063249674,0.0008592838,0.00008981605,0.000014587074,0.00013853655,0.0004623733,0.001271545,0.00052901113,0.9599105,0.036594544,0.000036567406],"about_ca_topic_score_codex":0.0026529308,"about_ca_topic_score_gemma":0.0012566312,"teacher_disagreement_score":0.02507167,"about_ca_system_score_codex":0.0021703453,"about_ca_system_score_gemma":0.001839876,"threshold_uncertainty_score":0.08387315},"labels":[],"label_agreement":null},{"id":"W2904453761","doi":"10.48550/arxiv.1812.02900","title":"Off-Policy Deep Reinforcement Learning without Exploration","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":280,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Extrapolation; Artificial intelligence; Reinforcement; Class (philosophy); Uncorrelated; Space (punctuation); Action (physics); Control (management); Machine learning; Mathematics; Engineering","score_opus":0.0781399192839883,"score_gpt":0.21755028897240594,"score_spread":0.13941036968841763,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2904453761","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06402618,0.0002984046,0.92975116,0.00042414755,0.00007065608,0.000083649385,0.000058540423,0.000870413,0.0044169044],"genre_scores_gemma":[0.93641347,0.000104217674,0.06045766,0.00020281054,0.000022682196,0.00012467752,0.000072874616,0.000054399774,0.002547248],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99948657,0.00018096306,0.00002546252,0.000117328156,0.00010374634,0.000085875305],"domain_scores_gemma":[0.9972153,0.0018673309,0.00022650528,0.00031401773,0.00024332794,0.00013355695],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017545902,0.00087791606,0.0009195449,0.00023585884,0.0002485543,0.00063789304,0.0012756342,0.0009281942,0.0017959757],"category_scores_gemma":[0.0065652663,0.00038391916,0.0002757785,0.00026476188,0.0016031341,0.0011564806,0.0010974711,0.0017110857,0.0003052453],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019618332,0.00010252645,0.0009824656,0.00007172533,0.000029393768,0.000059453592,0.000057946567,0.92635995,0.0016647131,0.018604768,0.0012206528,0.050650306],"study_design_scores_gemma":[0.000013894271,0.00003694152,0.00005309807,0.0000049236896,0.0000024814715,0.0000071974473,0.0000036024578,0.9935582,0.00044940464,0.005643531,0.0002238565,0.0000029253135],"about_ca_topic_score_codex":0.0040813987,"about_ca_topic_score_gemma":0.0029695795,"teacher_disagreement_score":0.0040813987,"about_ca_system_score_codex":0.0009173156,"about_ca_system_score_gemma":0.001365539,"threshold_uncertainty_score":0.009279311},"labels":[],"label_agreement":null},{"id":"W2905224739","doi":"10.1609/aaai.v33i01.33014504","title":"A Comparative Analysis of Expected and Distributional Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":42,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Convergence (economics); Computer science; Mathematical optimization; Linear approximation; Mathematics; Econometrics; Artificial intelligence; Nonlinear system; Economics","score_opus":0.058978497480422894,"score_gpt":0.3011855220912152,"score_spread":0.2422070246107923,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2905224739","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.097379744,0.0023395962,0.88758695,0.0016484924,0.00008660953,0.00008967878,0.00007684207,0.0007308122,0.01006136],"genre_scores_gemma":[0.9278194,0.0005806717,0.06919786,0.0002579053,0.00007588551,0.00011633221,0.00014850679,0.00014280873,0.0016605781],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9894434,0.006190744,0.00037036103,0.0009737304,0.0025679776,0.00045372976],"domain_scores_gemma":[0.9177887,0.07003044,0.0029508695,0.004334801,0.0038644338,0.0010307854],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014447526,0.00065744127,0.0012662304,0.0010564373,0.0005262428,0.0018834448,0.0022812767,0.0015200785,0.0034739096],"category_scores_gemma":[0.08589091,0.00039141162,0.0007355409,0.0007650662,0.002676139,0.0050743623,0.00253669,0.0023354825,0.00034626943],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005763393,0.00030368948,0.0055265366,0.00042667397,0.00020851621,0.000094242765,0.00027666515,0.65047646,0.0013270318,0.24082002,0.0015621635,0.09840161],"study_design_scores_gemma":[0.000029721261,0.00023618437,0.00079997507,0.000039905557,0.000015306696,0.000040576135,0.000044723696,0.93788326,0.0004967547,0.059800636,0.0005973191,0.00001566181],"about_ca_topic_score_codex":0.0013619815,"about_ca_topic_score_gemma":0.0013651294,"teacher_disagreement_score":0.014447526,"about_ca_system_score_codex":0.0019103087,"about_ca_system_score_gemma":0.0014915059,"threshold_uncertainty_score":0.07640672},"labels":[],"label_agreement":null},{"id":"W2906796853","doi":"10.1109/ijcnn52387.2021.9533459","title":"Dynamic Planning Networks","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Ontario Centres of Excellence","keywords":"Traverse; Computer science; Generalization; Reinforcement learning; Action (physics); Artificial intelligence; Construct (python library); State (computer science); Architecture; Machine learning; Algorithm; Mathematics","score_opus":0.019736730170050912,"score_gpt":0.27609669820521315,"score_spread":0.25635996803516226,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2906796853","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0072185327,0.0006986477,0.9752435,0.0006491133,0.00012142194,0.00007029762,0.00070666266,0.0019219453,0.013369941],"genre_scores_gemma":[0.4856665,0.0017996705,0.49332407,0.0005656411,0.00010737099,0.00048134293,0.0024375424,0.00043027807,0.015187578],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99963033,0.000072514515,0.000021178412,0.00013573075,0.00009483552,0.000045349912],"domain_scores_gemma":[0.999403,0.00029177807,0.000068043075,0.00008306733,0.00009595556,0.00005826263],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004351236,0.0008209219,0.00047039622,0.0004982962,0.00041235975,0.001043901,0.001508759,0.0008146584,0.009008932],"category_scores_gemma":[0.0026942194,0.0004967757,0.00054079166,0.00051227084,0.0007862134,0.0016299294,0.00142165,0.001543708,0.0014185585],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000097763754,0.00006088529,0.0009111206,0.00022640104,0.00006268969,0.00015158302,0.00007999634,0.7026433,0.0030269413,0.11348433,0.0098673785,0.1693876],"study_design_scores_gemma":[0.0000126149225,0.000019106985,0.00009136895,0.000023516119,0.0000105778245,0.000038842838,0.000010180921,0.9119513,0.0009919013,0.07819947,0.008642909,0.000008221285],"about_ca_topic_score_codex":0.004619542,"about_ca_topic_score_gemma":0.0075765345,"teacher_disagreement_score":0.009008932,"about_ca_system_score_codex":0.001024061,"about_ca_system_score_gemma":0.0013927792,"threshold_uncertainty_score":0.030137897},"labels":[],"label_agreement":null},{"id":"W2908832955","doi":"10.1109/smc.2018.00226","title":"Solving Home Robotics Challenges with Game Theory and Machine Learning","year":2018,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Royal Military College of Canada","funders":"","keywords":"Learning automata; PID controller; Automaton; Robot; Computer science; Action (physics); Motion planning; Artificial intelligence; Path (computing); Finite-state machine; Automata theory; Controller (irrigation); Robotics; Game theory; Mobile robot; Control engineering; Control (management); Engineering; Temperature control; Mathematics; Algorithm","score_opus":0.022491271114616428,"score_gpt":0.22942792053014177,"score_spread":0.20693664941552534,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2908832955","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011715394,0.00064092845,0.9826315,0.0009316307,0.000055977536,0.00003409099,0.000015209592,0.00011397048,0.0038612806],"genre_scores_gemma":[0.68826675,0.0013928671,0.3052107,0.0004325551,0.000153603,0.00025869624,0.00006369743,0.00007740109,0.004143725],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99914575,0.00044927798,0.000037178295,0.00011894297,0.00017723042,0.00007165457],"domain_scores_gemma":[0.9978709,0.0016870076,0.00009386034,0.0001271974,0.0001593237,0.00006172335],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015199606,0.00063744426,0.0011577521,0.00041018947,0.00066769426,0.0014622457,0.0012874606,0.0015993128,0.0017697951],"category_scores_gemma":[0.00408596,0.00049616885,0.00075058104,0.00036808892,0.0023901921,0.002231645,0.0015844775,0.0022153808,0.0002458411],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000047974816,0.00008372152,0.0004653523,0.00020280994,0.000075513424,0.00010175095,0.00021069759,0.75954443,0.0013119419,0.18729788,0.0016609704,0.04899692],"study_design_scores_gemma":[0.000012430214,0.000026820393,0.00006527904,0.000013491442,0.0000058759138,0.000019397592,0.000049268667,0.8785052,0.00032922564,0.11942034,0.0015440137,0.000008595669],"about_ca_topic_score_codex":0.004668828,"about_ca_topic_score_gemma":0.003381491,"teacher_disagreement_score":0.004668828,"about_ca_system_score_codex":0.0011546275,"about_ca_system_score_gemma":0.0013690947,"threshold_uncertainty_score":0.009283304},"labels":[],"label_agreement":null},{"id":"W2909958171","doi":"10.1007/978-3-030-10546-4_2","title":"Reinforcement Learning and Deep Reinforcement Learning","year":2019,"lang":"en","type":"book-chapter","venue":"Springer briefs in electrical and computer engineering","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Reinforcement learning; Reinforcement; Artificial intelligence; Deep learning; Q-learning; Computer science; Psychology; Social psychology","score_opus":0.006681597574720062,"score_gpt":0.19333488369683516,"score_spread":0.1866532861221151,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2909958171","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0045198533,0.17592363,0.59903526,0.006806851,0.006217887,0.00004696691,0.00039583855,0.0010269402,0.20602684],"genre_scores_gemma":[0.17973134,0.15179028,0.14878735,0.0020044628,0.006532528,0.00020169075,0.00088858895,0.0006360666,0.5094278],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99982375,0.00003747708,0.000009783261,0.000038132224,0.00007467444,0.000016214575],"domain_scores_gemma":[0.9997161,0.00016856479,0.000020162513,0.00003367671,0.000043106993,0.000018397297],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003208279,0.00089521846,0.0006922636,0.0005892866,0.00017922952,0.0012388392,0.0005825323,0.000923981,0.015370392],"category_scores_gemma":[0.0012203058,0.00034257566,0.00027534316,0.0011705947,0.0009525597,0.001815763,0.000710169,0.0019786628,0.004113698],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003408728,0.00006126606,0.00014280871,0.00043911062,0.000025646152,0.000048420305,0.0000444106,0.037078038,0.00123445,0.28321126,0.064644516,0.613036],"study_design_scores_gemma":[0.000013150101,0.000053460444,0.00048367548,0.00032813108,0.000018288982,0.00015833735,0.000032285185,0.09194309,0.0019483304,0.55302423,0.35196033,0.000036660465],"about_ca_topic_score_codex":0.0012121663,"about_ca_topic_score_gemma":0.0015545505,"teacher_disagreement_score":0.015370392,"about_ca_system_score_codex":0.0009308565,"about_ca_system_score_gemma":0.0005602767,"threshold_uncertainty_score":0.05141908},"labels":[],"label_agreement":null},{"id":"W291243768","doi":"","title":"Efficient Reinforcement Learning with Multiple Reward Functions for Randomized Controlled Trial Analysis","year":2010,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Generalization; Set (abstract data type); Extension (predicate logic); Reinforcement; State (computer science); Artificial intelligence; Mathematical optimization; Mathematics; Algorithm; Psychology","score_opus":0.00903513300406085,"score_gpt":0.24007088710621052,"score_spread":0.23103575410214966,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W291243768","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005676456,0.00007654783,0.99881786,0.000041631592,0.0000109865905,0.00003855356,0.000009166466,0.00014954586,0.00028808185],"genre_scores_gemma":[0.104506515,0.00021011187,0.8926597,0.00011466163,0.000067789086,0.00067691144,0.00008051581,0.00020700299,0.001476795],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99409217,0.003303097,0.0003287348,0.00067576236,0.0012818368,0.0003183832],"domain_scores_gemma":[0.9870085,0.010042634,0.0007515086,0.001035267,0.0008574819,0.00030463585],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.007460658,0.0016285557,0.0025316568,0.00096277654,0.0005114351,0.0016449976,0.0028341329,0.0016868856,0.0049599856],"category_scores_gemma":[0.029387495,0.0010006222,0.0011306938,0.001037513,0.0021283429,0.0029764487,0.0024828212,0.0041337893,0.0009796185],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002769802,0.00016925616,0.0005733698,0.00024428454,0.00011418941,0.00011023964,0.000095091076,0.6730787,0.0023652576,0.18418574,0.0023756633,0.13641128],"study_design_scores_gemma":[0.000056745324,0.00004172439,0.00004110741,0.00001678019,0.000010463208,0.000016722934,0.0000027973867,0.9508876,0.0005665319,0.04771642,0.0006319138,0.000011128026],"about_ca_topic_score_codex":0.0020917433,"about_ca_topic_score_gemma":0.0024927373,"teacher_disagreement_score":0.99253935,"about_ca_system_score_codex":0.002062509,"about_ca_system_score_gemma":0.0027148523,"threshold_uncertainty_score":0.03945619},"labels":[],"label_agreement":null},{"id":"W2912943342","doi":"","title":"Improving generalization in reinforcement learning on Atari 2600 games","year":2019,"lang":"en","type":"article","venue":"International journal of advance research, ideas and innovations in technology","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Hyperparameter; Machine learning; Regularization (linguistics); Overfitting; Deep learning; Transfer of learning; Artificial neural network","score_opus":0.01771163940650869,"score_gpt":0.3392445821240863,"score_spread":0.3215329427175776,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2912943342","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6366027,0.00045193636,0.34852427,0.0010112011,0.00016662558,0.00023145994,0.0001823757,0.0014405184,0.011388849],"genre_scores_gemma":[0.9597404,0.00007868638,0.036661472,0.00021408581,0.000017881646,0.00009698206,0.00016156548,0.00006113211,0.0029677912],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994803,0.00022083225,0.00002480809,0.00010500686,0.00008168199,0.00008741313],"domain_scores_gemma":[0.9980896,0.0012065296,0.00011998978,0.00018106705,0.00025257817,0.00015029407],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018450656,0.0009495034,0.00091895857,0.00032290717,0.00038719733,0.00059415563,0.0012717121,0.0007998252,0.0021097802],"category_scores_gemma":[0.007616494,0.0003402559,0.00054459984,0.0001716334,0.0009679478,0.001168163,0.0013615639,0.0020721606,0.000375787],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014405664,0.00020077128,0.0016721728,0.000051674364,0.00003496259,0.000072953546,0.000097336124,0.9633986,0.0014193711,0.0049407184,0.0011001547,0.026867216],"study_design_scores_gemma":[0.000010735486,0.000054094715,0.00015331477,0.000004251443,0.0000024343422,0.000004066214,0.0000082298175,0.99721515,0.00028297552,0.0021054782,0.00015631077,0.000002848823],"about_ca_topic_score_codex":0.010699531,"about_ca_topic_score_gemma":0.009996091,"teacher_disagreement_score":0.010699531,"about_ca_system_score_codex":0.0012028929,"about_ca_system_score_gemma":0.0009031365,"threshold_uncertainty_score":0.021274507},"labels":[],"label_agreement":null},{"id":"W2913966743","doi":"10.1109/cdc.2018.8619157","title":"On Passivity and Reinforcement Learning in Finite Games","year":2018,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Passivity; Reinforcement learning; Convergence (economics); Monotonic function; Stochastic game; Computer science; Property (philosophy); Class (philosophy); Exploit; Reinforcement; Mathematical optimization; Scheme (mathematics); Artificial intelligence; Mathematics; Mathematical economics; Engineering; Economics","score_opus":0.014901823723644646,"score_gpt":0.2495275137318254,"score_spread":0.23462569000818076,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2913966743","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005781117,0.00022824608,0.98929065,0.0001890473,0.000035326553,0.00004455519,0.000016186295,0.000055611355,0.0043592337],"genre_scores_gemma":[0.8148684,0.0013054389,0.17345917,0.00037096554,0.0001984759,0.00063220644,0.00008543183,0.00013529255,0.008944636],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9981502,0.00086884684,0.000113370006,0.0002776372,0.00040771326,0.00018228572],"domain_scores_gemma":[0.9927503,0.005452305,0.0005216072,0.00042238378,0.00056751235,0.00028588733],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045321197,0.0016072042,0.001256049,0.00089241023,0.000577748,0.001581156,0.0015794383,0.001417225,0.0032203621],"category_scores_gemma":[0.011792503,0.0005990372,0.0017371187,0.00049008225,0.0043514646,0.002299678,0.0022886067,0.002993115,0.00042439238],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005075273,0.000053462514,0.00028684773,0.00015472784,0.00007531101,0.00012827363,0.00021004787,0.39077666,0.003315719,0.5924007,0.00041557965,0.012131867],"study_design_scores_gemma":[0.000023131457,0.00012273203,0.00010381661,0.00004082768,0.00002008054,0.00003627651,0.000013052981,0.74021316,0.0010876018,0.25688416,0.0014346496,0.00002047507],"about_ca_topic_score_codex":0.0027291235,"about_ca_topic_score_gemma":0.0015509473,"teacher_disagreement_score":0.0045321197,"about_ca_system_score_codex":0.0022164772,"about_ca_system_score_gemma":0.0017291838,"threshold_uncertainty_score":0.023968399},"labels":[],"label_agreement":null},{"id":"W2914461813","doi":"10.1007/978-3-030-04735-1_3","title":"Emergent Policy Discovery for Visual Reinforcement Learning Through Tangled Program Graphs: A Tutorial","year":2019,"lang":"en","type":"book-chapter","venue":"Genetic and evolutionary computation","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Constructive; Task (project management); Computer science; A priori and a posteriori; Artificial intelligence; Human–computer interaction; Theoretical computer science; Machine learning; Programming language; Engineering","score_opus":0.016988921752373022,"score_gpt":0.27830949845595915,"score_spread":0.2613205767035861,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2914461813","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002332754,0.016537555,0.9564329,0.0007305812,0.00030665036,0.000052222887,0.00010654241,0.00092223106,0.02257854],"genre_scores_gemma":[0.12255886,0.04029019,0.7801695,0.00067904795,0.00083356304,0.00041803665,0.0006077805,0.0008942941,0.0535487],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99987733,0.00002592696,0.00000805514,0.00003538257,0.000042990443,0.000010235889],"domain_scores_gemma":[0.9997496,0.00017632973,0.000013409029,0.000020063942,0.000025886691,0.000014625395],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00039182042,0.0010079714,0.0007334547,0.00066589774,0.00024918857,0.0010316063,0.0009416363,0.00093257957,0.00879219],"category_scores_gemma":[0.0009971942,0.0005757092,0.0009060308,0.0009162024,0.00073747634,0.0015252525,0.0011221071,0.002055846,0.0017651006],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000040138846,0.00010640317,0.00022930649,0.00087815104,0.00009474846,0.00016637851,0.00015960612,0.17460884,0.005326238,0.28232485,0.028355189,0.50771004],"study_design_scores_gemma":[0.000018406145,0.000051221836,0.0002563164,0.00022471839,0.000025591155,0.00021216975,0.000033162607,0.44729307,0.0022029052,0.4828451,0.06679603,0.000041290827],"about_ca_topic_score_codex":0.0011014149,"about_ca_topic_score_gemma":0.001697943,"teacher_disagreement_score":0.00879219,"about_ca_system_score_codex":0.00085571676,"about_ca_system_score_gemma":0.0004071941,"threshold_uncertainty_score":0.029412806},"labels":[],"label_agreement":null},{"id":"W2918549499","doi":"10.1145/3319619.3321956","title":"Novelty search for deep reinforcement learning policy network weights by action sequence edit metric distance","year":2019,"lang":"en","type":"preprint","venue":"Proceedings of the Genetic and Evolutionary Computation Conference Companion","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Reinforcement learning; Artificial intelligence; Neuroevolution; Computer science; Novelty; Metric (unit); Benchmark (surveying); Deep learning; Machine learning; Action (physics); Artificial neural network","score_opus":0.04233092629446205,"score_gpt":0.2901398948073653,"score_spread":0.24780896851290324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2918549499","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05265215,0.00026889023,0.9440467,0.00023920563,0.00004959806,0.00007083257,0.000041447962,0.00040105163,0.0022302032],"genre_scores_gemma":[0.7765415,0.00011654275,0.22031596,0.0001416636,0.00003849992,0.00020844363,0.00009548179,0.00010854447,0.0024332942],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99946994,0.00017945816,0.000028283303,0.00010845303,0.00016861252,0.000045371275],"domain_scores_gemma":[0.99836475,0.0009832468,0.00018592109,0.00011279255,0.0002606433,0.00009272525],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017914847,0.0006706649,0.0008712598,0.0007377585,0.00038777394,0.00068279565,0.0012806851,0.0012047701,0.0015909745],"category_scores_gemma":[0.0056789075,0.00035103344,0.00043485677,0.00047317904,0.0009230227,0.0010026614,0.0011317628,0.001300584,0.00024489948],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007628121,0.00008469723,0.0011831842,0.000050239054,0.000042877382,0.00005467124,0.00005030574,0.91432077,0.0025752545,0.012185635,0.0008320245,0.06854406],"study_design_scores_gemma":[0.00000650185,0.00002532487,0.000061511426,0.000002577189,0.0000024455642,0.0000088292345,0.000001958544,0.9969296,0.00032669137,0.0024929964,0.00013914115,0.0000024454405],"about_ca_topic_score_codex":0.0029064242,"about_ca_topic_score_gemma":0.0027464058,"teacher_disagreement_score":0.0029064242,"about_ca_system_score_codex":0.0015754821,"about_ca_system_score_gemma":0.0010095331,"threshold_uncertainty_score":0.011430979},"labels":[],"label_agreement":null},{"id":"W2919205453","doi":"10.48550/arxiv.1903.00194","title":"Should All Temporal Difference Learning Use Emphasis?","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Counterexample; Convergence (economics); Temporal difference learning; Computer science; Simple (philosophy); Class (philosophy); Artificial intelligence; Mathematics; Economics; Reinforcement learning; Epistemology; Discrete mathematics","score_opus":0.18257296351319344,"score_gpt":0.22286829858392074,"score_spread":0.0402953350707273,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2919205453","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02381364,0.0050856983,0.90697765,0.033077452,0.0018896355,0.000111174704,0.0001230147,0.00081486185,0.028106824],"genre_scores_gemma":[0.6607624,0.0033114534,0.2982968,0.010682047,0.0013835189,0.00029773067,0.0002190515,0.0005500684,0.024496892],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983119,0.0007648166,0.000085919615,0.0004110124,0.00030389652,0.00012246873],"domain_scores_gemma":[0.99340206,0.004153691,0.00024198153,0.0010408652,0.0008194866,0.00034191957],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0060481583,0.00074273505,0.0010662656,0.00035885378,0.0007097326,0.0020264285,0.0022673232,0.0026897755,0.0075709685],"category_scores_gemma":[0.02751793,0.00036277625,0.00049359066,0.0005453008,0.002706529,0.0078382585,0.0020388803,0.004984076,0.002029027],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00050344213,0.000225499,0.0032383006,0.00045624157,0.00012734055,0.00012215556,0.00035972762,0.041024692,0.0034659528,0.49059907,0.018872889,0.44100478],"study_design_scores_gemma":[0.00015367458,0.00037871586,0.00087983697,0.0002447818,0.000037596525,0.0001819056,0.00018291423,0.26785904,0.0046411487,0.6900384,0.035352144,0.00004994087],"about_ca_topic_score_codex":0.0023320685,"about_ca_topic_score_gemma":0.0021284188,"teacher_disagreement_score":0.0075709685,"about_ca_system_score_codex":0.001119188,"about_ca_system_score_gemma":0.0013085328,"threshold_uncertainty_score":0.031986117},"labels":[],"label_agreement":null},{"id":"W2922221969","doi":"10.65109/knvj7743","title":"On the Pitfalls of Measuring Emergent Communication","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Computer science; Set (abstract data type); Task (project management); Action (physics); Human–computer interaction; Reinforcement learning; Channel (broadcasting); Simple (philosophy); Measure (data warehouse); Data science; Artificial intelligence; Telecommunications; Data mining; Engineering","score_opus":0.05880800494805078,"score_gpt":0.26068876717904227,"score_spread":0.2018807622309915,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2922221969","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09248709,0.0057318206,0.884335,0.0058176294,0.00031520895,0.00017157751,0.00041945314,0.0012079629,0.009514188],"genre_scores_gemma":[0.88509196,0.0014180889,0.11088207,0.0006988764,0.00019219532,0.00034987435,0.0003032453,0.00027164124,0.0007919488],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9757992,0.014615609,0.0015784985,0.0027415196,0.0045972057,0.00066788384],"domain_scores_gemma":[0.7547623,0.20330405,0.01288074,0.014574119,0.011833345,0.0026453263],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.019992508,0.0018598284,0.00210862,0.0035746288,0.001637107,0.003889646,0.0027511362,0.0030018834,0.0019083278],"category_scores_gemma":[0.19622728,0.00085459714,0.00091399654,0.0028675392,0.009355364,0.015154216,0.004727256,0.0061971704,0.00060374534],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000681354,0.00037624143,0.042348474,0.0020456957,0.00063246686,0.00034823737,0.0041190526,0.3201394,0.0068017333,0.30999693,0.0072842375,0.3052262],"study_design_scores_gemma":[0.000049207607,0.00029906712,0.00776275,0.0003333679,0.0000596157,0.00030078343,0.0010329438,0.39508045,0.0045879395,0.58546776,0.0048385616,0.00018758187],"about_ca_topic_score_codex":0.003908347,"about_ca_topic_score_gemma":0.0023956378,"teacher_disagreement_score":0.98000747,"about_ca_system_score_codex":0.0023840545,"about_ca_system_score_gemma":0.0016220942,"threshold_uncertainty_score":0.105731785},"labels":[],"label_agreement":null},{"id":"W2923023063","doi":"","title":"Modeling the Long Term Future in Model-Based Reinforcement Learning","year":2018,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Polytechnique Montréal","funders":"","keywords":"Reinforcement learning; Term (time); Computer science; Artificial intelligence","score_opus":0.05336023462993137,"score_gpt":0.3395181485785783,"score_spread":0.2861579139486469,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2923023063","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08642178,0.0007325049,0.90768874,0.0010022697,0.00009572246,0.00003256849,0.00016907745,0.00024714464,0.0036101865],"genre_scores_gemma":[0.96633667,0.0003112814,0.030400256,0.00008139247,0.0000318125,0.00008269882,0.00010697818,0.000032166103,0.0026167687],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995641,0.00020412485,0.000020877858,0.000093699426,0.000060948645,0.00005635479],"domain_scores_gemma":[0.99785197,0.0015201746,0.00021298896,0.00010464161,0.00019532182,0.0001148597],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015156094,0.0005987287,0.00095210056,0.00042878193,0.00035371535,0.0012816265,0.0012628008,0.0011486515,0.0026144835],"category_scores_gemma":[0.006533797,0.0005099331,0.00045607673,0.00051269826,0.0010309039,0.0023575402,0.0010672596,0.0018679508,0.0002447881],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004797834,0.000023015717,0.00052175875,0.000029881026,0.000020905167,0.00003467367,0.000042402928,0.9583411,0.00020896866,0.032673415,0.00031934158,0.007736503],"study_design_scores_gemma":[0.000006006694,0.000009444351,0.00003693736,0.0000034463158,0.000004091029,0.000004182461,0.00000364058,0.98543996,0.000039621653,0.01435796,0.00009193231,0.000002745026],"about_ca_topic_score_codex":0.0086563965,"about_ca_topic_score_gemma":0.008916649,"teacher_disagreement_score":0.0086563965,"about_ca_system_score_codex":0.0012342923,"about_ca_system_score_gemma":0.000968948,"threshold_uncertainty_score":0.017211974},"labels":[],"label_agreement":null},{"id":"W2924816077","doi":"10.1613/jair.1.11418","title":"Modeling and Planning with Macro-Actions in Decentralized POMDPs","year":2019,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":62,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Office of Naval Research; National Institute of Mental Health; Multidisciplinary University Research Initiative; Air Force Office of Scientific Research; Defense Advanced Research Projects Agency; National Institutes of Health; National Science Foundation","keywords":"Computer science; Partially observable Markov decision process; Macro; Markov decision process; Exploit; Action (physics); Class (philosophy); Mathematical optimization; Artificial intelligence; Markov chain; Markov process; Markov model; Machine learning","score_opus":0.20964206851014752,"score_gpt":0.4311779567163518,"score_spread":0.22153588820620426,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2924816077","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020108098,0.0001562998,0.97708225,0.00015275179,0.000021293994,0.000055647615,0.00012351862,0.00028625564,0.0020138917],"genre_scores_gemma":[0.67246115,0.0004029033,0.32424262,0.000093236595,0.000035010406,0.00034660927,0.00027133565,0.00007795761,0.0020691608],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993474,0.0002297257,0.000044490043,0.00014457383,0.00014822533,0.00008562833],"domain_scores_gemma":[0.9986859,0.0008383844,0.00017907865,0.00011758889,0.00010250467,0.00007652195],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001276598,0.0007442669,0.0007625017,0.00037669926,0.00052034564,0.0011725494,0.0010290835,0.0008020538,0.001733584],"category_scores_gemma":[0.0029430045,0.0005720668,0.0007816371,0.00051418215,0.0012037026,0.0012813654,0.0013460508,0.0015182246,0.00020224067],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000022989452,0.000012561754,0.00029191346,0.00003246828,0.000009994695,0.000047077545,0.000037343758,0.9762113,0.00037265813,0.017381912,0.0001785192,0.0054012984],"study_design_scores_gemma":[0.000009949988,0.000011027895,0.00006496305,0.000004597137,0.000003593588,0.000007650508,0.000010849337,0.98498523,0.00023766533,0.014133676,0.00052763923,0.0000031986551],"about_ca_topic_score_codex":0.0057927417,"about_ca_topic_score_gemma":0.0073644486,"teacher_disagreement_score":0.0057927417,"about_ca_system_score_codex":0.0010735804,"about_ca_system_score_gemma":0.0013823872,"threshold_uncertainty_score":0.011518061},"labels":[],"label_agreement":null},{"id":"W2925216817","doi":"10.48550/arxiv.1903.09295","title":"DQN with model-based exploration: efficient learning on environments with sparse rewards","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Artificial intelligence; Machine learning","score_opus":0.06704040110681203,"score_gpt":0.17622591528041712,"score_spread":0.10918551417360509,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2925216817","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04778836,0.00024470093,0.9481751,0.00049831846,0.000049700124,0.000060770235,0.00008672336,0.0008272582,0.0022691519],"genre_scores_gemma":[0.7587944,0.00019297066,0.23659986,0.0004387686,0.00004115021,0.00021196611,0.00025355283,0.00013143138,0.0033359036],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995952,0.00011998538,0.000018414627,0.00011463244,0.000078391495,0.00007325618],"domain_scores_gemma":[0.998868,0.00063062273,0.00011309127,0.00013123784,0.00014010543,0.0001169103],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001094018,0.0008782345,0.0010698457,0.00040748628,0.00042791342,0.00075489003,0.0021460503,0.0012530927,0.0022982901],"category_scores_gemma":[0.003999881,0.00071773387,0.000554791,0.00048601284,0.0010812655,0.0019370817,0.0025099637,0.0017823695,0.00036348336],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010818869,0.00008681275,0.0013228436,0.000063555075,0.000032047014,0.00007861435,0.00007493197,0.9262824,0.0014969168,0.009825586,0.0012906746,0.05933737],"study_design_scores_gemma":[0.000011633678,0.000017777345,0.00003805982,0.0000029716464,0.0000022242111,0.0000069655143,0.000003142296,0.9952395,0.0002182629,0.004276354,0.00018092094,0.0000021572025],"about_ca_topic_score_codex":0.0065501374,"about_ca_topic_score_gemma":0.0068267477,"teacher_disagreement_score":0.0065501374,"about_ca_system_score_codex":0.000939822,"about_ca_system_score_gemma":0.0015011473,"threshold_uncertainty_score":0.013024032},"labels":[],"label_agreement":null},{"id":"W2925234320","doi":"10.48550/arxiv.1903.09762","title":"TTR-Based Reward for Reinforcement Learning with Implicit Model Priors","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Reinforcement learning; Computer science; Key (lock); Function (biology); Inefficiency; Artificial intelligence; Process (computing); Exploit; Temporal difference learning; State (computer science); Machine learning; Algorithm","score_opus":0.06577840482045562,"score_gpt":0.19876942502459402,"score_spread":0.1329910202041384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2925234320","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01160518,0.00014767493,0.9858348,0.00014888795,0.00002919312,0.000041826064,0.000036749167,0.00061107543,0.0015446009],"genre_scores_gemma":[0.8449294,0.00014179458,0.15177546,0.00017772174,0.00003610402,0.00019513535,0.0001428289,0.00018438027,0.002417254],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990853,0.00034826432,0.000047405807,0.000171101,0.00023813132,0.00010981377],"domain_scores_gemma":[0.996606,0.0023604701,0.00029476936,0.0002937109,0.00031786534,0.00012713543],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020174982,0.0011432873,0.0013351713,0.0005121175,0.0003997674,0.00087109994,0.0016440097,0.0011825816,0.0032228825],"category_scores_gemma":[0.009917859,0.0004597312,0.000568279,0.00047639257,0.0013597762,0.0016857481,0.0013569439,0.0020201872,0.0005051074],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000076896904,0.00005702032,0.00029256957,0.000049776187,0.000018199187,0.000033300694,0.000027467926,0.96133345,0.00073228485,0.013650082,0.00064688904,0.02308215],"study_design_scores_gemma":[0.0000067049623,0.000014087423,0.000021601156,0.0000026078328,0.0000021256496,0.0000040514783,0.0000010432659,0.9961392,0.00019008935,0.0035211656,0.000094759,0.0000024509563],"about_ca_topic_score_codex":0.004499862,"about_ca_topic_score_gemma":0.003702425,"teacher_disagreement_score":0.004499862,"about_ca_system_score_codex":0.0015241064,"about_ca_system_score_gemma":0.0019006929,"threshold_uncertainty_score":0.011058211},"labels":[],"label_agreement":null},{"id":"W2928034659","doi":"10.1109/icpr48806.2021.9412212","title":"Deep Reinforcement Learning on a Budget: 3D Control and Reasoning Without a Supercomputer","year":2021,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; College of Natural Resources and Sciences, Humboldt State University; Agence Nationale de la Recherche","keywords":"Computer science; Reinforcement learning; Artificial intelligence; Robotics; High fidelity; Deep learning; Suite; Human–computer interaction; Fidelity; Distributed computing; Machine learning; Robot","score_opus":0.008855926514395013,"score_gpt":0.23073166799091988,"score_spread":0.22187574147652486,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2928034659","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.055938877,0.00024045868,0.9287863,0.0007567216,0.00013004905,0.00008302424,0.0002944813,0.0038260105,0.009944067],"genre_scores_gemma":[0.76025987,0.00019023132,0.23265156,0.00024905335,0.000033445445,0.00021275497,0.0006492148,0.00033216187,0.0054216515],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997222,0.00007716454,0.000013771245,0.000064797736,0.000079804726,0.000042182226],"domain_scores_gemma":[0.9994735,0.00022655624,0.000033701628,0.0001314407,0.00007165729,0.00006313646],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007547914,0.00059213344,0.0005705364,0.000218121,0.00038265486,0.00077196857,0.0014742434,0.0008497653,0.007778412],"category_scores_gemma":[0.0024056952,0.0004716442,0.0004954292,0.0002983988,0.000975573,0.0014472138,0.001593856,0.0016435034,0.0010157978],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014478179,0.00004867008,0.00036041668,0.000034771198,0.000018119994,0.000058782392,0.00003366125,0.95346683,0.0019141414,0.009209968,0.0028917808,0.031818025],"study_design_scores_gemma":[0.00001015202,0.000012788408,0.000040984338,0.0000024465292,0.0000015811775,0.000003549628,0.0000028135246,0.9959363,0.00042508144,0.0030484989,0.00051404996,0.0000017464274],"about_ca_topic_score_codex":0.012484902,"about_ca_topic_score_gemma":0.013154619,"teacher_disagreement_score":0.012484902,"about_ca_system_score_codex":0.0011391449,"about_ca_system_score_gemma":0.00095420115,"threshold_uncertainty_score":0.026021361},"labels":[],"label_agreement":null},{"id":"W2930281309","doi":"10.48550/arxiv.1904.01191","title":"Planning with Expectation Models","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Parametrization (atmospheric modeling); Convergence (economics); Mathematical optimization; Bellman equation; Function (biology); State (computer science); Artificial intelligence; Mathematics; Algorithm","score_opus":0.10319231109665329,"score_gpt":0.19053988156606383,"score_spread":0.08734757046941054,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2930281309","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00344733,0.00014491672,0.9938188,0.00024483644,0.00001880178,0.00002733439,0.00006999699,0.0002338011,0.001994122],"genre_scores_gemma":[0.6184328,0.0007229415,0.37194806,0.00042594605,0.00009969672,0.00048671782,0.0005383944,0.0002256095,0.0071198754],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99799734,0.0009421583,0.00010731767,0.0003961289,0.00039142463,0.0001656058],"domain_scores_gemma":[0.9955348,0.0034367717,0.00029495533,0.00034285613,0.0002635298,0.00012710637],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020229118,0.00124115,0.0012819732,0.0005611754,0.00045726646,0.0017985567,0.0019114218,0.0015321041,0.0051650256],"category_scores_gemma":[0.010590865,0.0006947867,0.0011963659,0.0008606566,0.0015623099,0.00343956,0.0020522566,0.0028029336,0.0008192444],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010469685,0.00005154536,0.0004843133,0.00011573069,0.000041152925,0.00010020156,0.00011561916,0.76591367,0.00054763135,0.19785133,0.0015817666,0.033092383],"study_design_scores_gemma":[0.00001684838,0.000029862225,0.00004717113,0.000014308227,0.0000086219925,0.000024583356,0.000010101476,0.90313154,0.0002960206,0.0954311,0.0009804006,0.000009461581],"about_ca_topic_score_codex":0.004583531,"about_ca_topic_score_gemma":0.0047242776,"teacher_disagreement_score":0.0051650256,"about_ca_system_score_codex":0.001479037,"about_ca_system_score_gemma":0.0016722215,"threshold_uncertainty_score":0.017278671},"labels":[],"label_agreement":null},{"id":"W2936568671","doi":"10.48550/arxiv.1904.09024","title":"When is a Prediction Knowledge?","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Variety (cybernetics); Computer science; Reinforcement learning; Value (mathematics); Artificial intelligence; Predictive value; Body of knowledge; Work (physics); Knowledge management; Data science; Epistemology; Machine learning; Engineering","score_opus":0.07118221310828741,"score_gpt":0.19075482663090618,"score_spread":0.11957261352261876,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2936568671","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.059608825,0.002999315,0.8134291,0.07229773,0.00064690964,0.000107416665,0.00078380195,0.000628553,0.04949832],"genre_scores_gemma":[0.93003315,0.0012653051,0.06305134,0.0020236722,0.00040168766,0.00011254376,0.00036687384,0.00015397527,0.0025915287],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9931514,0.0028704244,0.00042699123,0.00154784,0.001465302,0.0005378742],"domain_scores_gemma":[0.95791805,0.030154346,0.0022258232,0.0048162634,0.0038668113,0.0010187569],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012130862,0.00058457383,0.0012462212,0.0015981832,0.0019213868,0.0062477984,0.0021000768,0.0043627345,0.006528859],"category_scores_gemma":[0.056767438,0.00063052704,0.0009384208,0.0013766879,0.01347732,0.026561003,0.0029413484,0.005822073,0.0012756378],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001267474,0.0000620011,0.0029740664,0.00023745245,0.00007059238,0.00014827106,0.0011567506,0.009976132,0.00065231026,0.92500395,0.0030770877,0.056514606],"study_design_scores_gemma":[0.000010948865,0.000014448229,0.00035657804,0.00007676152,0.000013504476,0.000033124914,0.00029845678,0.011073739,0.0005007421,0.9838665,0.003737232,0.000018106402],"about_ca_topic_score_codex":0.003705441,"about_ca_topic_score_gemma":0.002782183,"teacher_disagreement_score":0.012130862,"about_ca_system_score_codex":0.002833319,"about_ca_system_score_gemma":0.0032468445,"threshold_uncertainty_score":0.06415486},"labels":[],"label_agreement":null},{"id":"W2941011530","doi":"","title":"Reward Estimation for Variance Reduction in Deep Reinforcement Learning","year":2018,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Variance reduction; Reduction (mathematics); Computer science; Variance (accounting); Estimation; Artificial intelligence; Reinforcement; Machine learning; Statistics; Econometrics; Psychology; Mathematics; Engineering; Economics; Social psychology","score_opus":0.050061910459263996,"score_gpt":0.3514664291660533,"score_spread":0.3014045187067893,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2941011530","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01240243,0.00022381908,0.9860772,0.00019185952,0.00004150278,0.000032927543,0.00003308846,0.00032204998,0.0006751623],"genre_scores_gemma":[0.8297433,0.00017739904,0.16629362,0.00021852128,0.000071182505,0.00020104398,0.00014918821,0.00019791274,0.0029478313],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99910563,0.00035871452,0.000049421677,0.00016258289,0.00019937176,0.00012420745],"domain_scores_gemma":[0.9969207,0.002177949,0.0001786062,0.00018059957,0.00043165236,0.00011055908],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025791873,0.00092977023,0.0015190844,0.00051068707,0.00036331706,0.00091552874,0.001321208,0.001285382,0.0025169428],"category_scores_gemma":[0.00972312,0.00074545795,0.0005431139,0.0004297151,0.000930149,0.0012221227,0.0016563904,0.0026558873,0.00037220013],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002384946,0.00015650017,0.0007556694,0.00011418948,0.000074379306,0.000040070594,0.00006098998,0.8548338,0.003151075,0.020109175,0.0020622534,0.11840348],"study_design_scores_gemma":[0.0000066961584,0.000014136576,0.000041486317,0.000004169227,0.0000033705323,0.0000030853294,0.0000013666319,0.995622,0.00029792025,0.0039321748,0.00007091211,0.000002669598],"about_ca_topic_score_codex":0.0036557694,"about_ca_topic_score_gemma":0.0041019507,"teacher_disagreement_score":0.0036557694,"about_ca_system_score_codex":0.0010921587,"about_ca_system_score_gemma":0.0015437132,"threshold_uncertainty_score":0.013640165},"labels":[],"label_agreement":null},{"id":"W2943366292","doi":"10.1038/s42256-019-0053-0","title":"Moving beyond reward prediction errors","year":2019,"lang":"en","type":"article","venue":"Nature Machine Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"The Scarborough Hospital; Canadian Institute for Advanced Research; University of Toronto; Vector Institute","funders":"","keywords":"Business; Computer science","score_opus":0.006224225736806918,"score_gpt":0.25221000272268734,"score_spread":0.24598577698588042,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2943366292","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029697219,0.0038162777,0.9216798,0.01391781,0.0011367259,0.000060968818,0.00019658335,0.00077535264,0.028719354],"genre_scores_gemma":[0.8509098,0.0018839687,0.11964226,0.0023332473,0.0007710717,0.0000915194,0.00019230814,0.00040625958,0.023769505],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.997806,0.0008681104,0.00008532252,0.00051686086,0.0005547218,0.00016894215],"domain_scores_gemma":[0.9813778,0.013502394,0.0008405723,0.0019179325,0.0017800755,0.0005811892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041875434,0.0013759801,0.0017420165,0.00083732215,0.0006879446,0.0026954534,0.0029007073,0.0030611386,0.012541482],"category_scores_gemma":[0.030998567,0.00083397265,0.00065376447,0.0006868813,0.0026849806,0.011031353,0.0027546776,0.007480556,0.0018285809],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031290736,0.00027262687,0.0026433014,0.00032275452,0.00019982058,0.00013849103,0.00018943763,0.2697171,0.0015442624,0.49796242,0.012494902,0.2142019],"study_design_scores_gemma":[0.000031600073,0.00006192661,0.00033836684,0.00006866611,0.000027700657,0.000025643778,0.000024603849,0.6198157,0.0006956669,0.37552816,0.0033601348,0.000021859338],"about_ca_topic_score_codex":0.003999275,"about_ca_topic_score_gemma":0.00308674,"teacher_disagreement_score":0.012541482,"about_ca_system_score_codex":0.001418739,"about_ca_system_score_gemma":0.0015398397,"threshold_uncertainty_score":0.04195541},"labels":[],"label_agreement":null},{"id":"W2945054955","doi":"10.48550/arxiv.1905.10016","title":"A Micro-Objective Perspective of Reinforcement Learning","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Perspective (graphical); Reinforcement learning; Reinforcement; Computer science; Psychology; Artificial intelligence; Social psychology","score_opus":0.04403034584286141,"score_gpt":0.1983845935839299,"score_spread":0.15435424774106848,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2945054955","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0036519305,0.0007814257,0.9836236,0.0010083198,0.00007174565,0.000027860508,0.00006633844,0.00007397567,0.010694711],"genre_scores_gemma":[0.65466946,0.0022697814,0.32850736,0.0005758462,0.00041813828,0.00027901446,0.00013045942,0.00013152802,0.0130183445],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99912804,0.00036398022,0.000037572492,0.00014694172,0.00025530075,0.00006831828],"domain_scores_gemma":[0.9987112,0.0006851837,0.00017547894,0.00014372185,0.0001816726,0.00010272228],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014484574,0.0010116753,0.00076147565,0.0005016118,0.00030789332,0.0019039051,0.0014160743,0.0009727547,0.0036119353],"category_scores_gemma":[0.002685951,0.00033008127,0.00072337146,0.000560226,0.002039669,0.0026840817,0.0010732873,0.0025535515,0.00047044264],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000021750882,0.000034191897,0.00030618787,0.00011593795,0.00004851406,0.00006367955,0.000089546775,0.2616112,0.0009731234,0.71761537,0.0010538794,0.018066613],"study_design_scores_gemma":[0.00001641119,0.000055361994,0.00017000473,0.0000391655,0.000018731283,0.000037868194,0.000031475138,0.5857909,0.0006147266,0.4073258,0.005883472,0.000016062075],"about_ca_topic_score_codex":0.002080925,"about_ca_topic_score_gemma":0.0021229004,"teacher_disagreement_score":0.0036119353,"about_ca_system_score_codex":0.0016265552,"about_ca_system_score_gemma":0.0010987887,"threshold_uncertainty_score":0.012083113},"labels":[],"label_agreement":null},{"id":"W2945090640","doi":"10.1007/978-3-030-18305-9_18","title":"Options in Multi-task Reinforcement Learning - Transfer via Reflection","year":2019,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Reinforcement learning; Regret; Computer science; Transfer of learning; Artificial intelligence; Context (archaeology); State space; Task (project management); Landmark; Formalism (music); Reflection (computer programming); Machine learning","score_opus":0.029580750455954798,"score_gpt":0.27283070230665457,"score_spread":0.24324995185069978,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2945090640","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015841028,0.00034253334,0.9777449,0.00021596519,0.00006177518,0.00004035713,0.000019127243,0.00032007202,0.005414331],"genre_scores_gemma":[0.83530307,0.00031103386,0.15409663,0.000116889925,0.000054677414,0.00026685654,0.000056363537,0.00015232532,0.0096421465],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99927527,0.00038489935,0.000037257054,0.00012974963,0.00009979532,0.000073024625],"domain_scores_gemma":[0.9979234,0.0016032956,0.00008574239,0.00018860116,0.00010274976,0.000096137606],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017537276,0.0007977373,0.0009882126,0.0002458922,0.00036327998,0.0010112262,0.0017285008,0.0014832788,0.0051956787],"category_scores_gemma":[0.0049606445,0.0004580801,0.0006955728,0.00035845494,0.0016799874,0.0024598283,0.002590037,0.0026562407,0.00063433766],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043197317,0.00016940126,0.00038163623,0.00020525898,0.00007878041,0.00013680251,0.00026381237,0.6049759,0.0037846216,0.21208343,0.0020005635,0.17548789],"study_design_scores_gemma":[0.000030956082,0.0000685684,0.00005591199,0.000013468148,0.000008358162,0.000018434197,0.000014628574,0.8880473,0.0007803703,0.11041296,0.0005381457,0.000010872069],"about_ca_topic_score_codex":0.0011674355,"about_ca_topic_score_gemma":0.00072219875,"teacher_disagreement_score":0.0051956787,"about_ca_system_score_codex":0.00065008475,"about_ca_system_score_gemma":0.0005616958,"threshold_uncertainty_score":0.01738125},"labels":[],"label_agreement":null},{"id":"W2945334994","doi":"10.65109/iadj1863","title":"Exploration in the Face of Parametric and Intrinsic Uncertainties","year":2019,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Alberta","funders":"","keywords":"Reinforcement learning; Parametric statistics; Schedule; Computer science; Face (sociological concept); Quantile; Mathematical optimization; Probability distribution; Artificial intelligence; Mathematics; Statistics","score_opus":0.024700165815132093,"score_gpt":0.24619408057435277,"score_spread":0.22149391475922067,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2945334994","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0806987,0.00025849382,0.91668105,0.00039494105,0.000027813316,0.000022783113,0.000036064368,0.00026813944,0.0016120141],"genre_scores_gemma":[0.95500684,0.00010554882,0.043325145,0.000104688486,0.000023617837,0.00004495216,0.000040057115,0.000047702426,0.0013015381],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991553,0.00037390128,0.000037273367,0.0001592928,0.00015052242,0.00012364233],"domain_scores_gemma":[0.99536395,0.0034981715,0.00034933208,0.00030892252,0.00028784917,0.00019175821],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021939457,0.00072389044,0.000900705,0.00032096496,0.00034207586,0.0007410305,0.0009779601,0.0007712778,0.0010264867],"category_scores_gemma":[0.008715289,0.0004678743,0.0004645313,0.00024838376,0.0015842824,0.0021105253,0.0022941334,0.0019755193,0.00016317598],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016729101,0.00003256998,0.000746357,0.000048992937,0.000027308,0.00007657732,0.00008728048,0.94795144,0.0022832153,0.015558718,0.0004925339,0.03252773],"study_design_scores_gemma":[0.000009764089,0.000023469202,0.00009956197,0.0000041511585,0.0000027889391,0.000014882082,0.00000836375,0.9857771,0.00048249503,0.013429113,0.0001434446,0.0000049040855],"about_ca_topic_score_codex":0.0020308392,"about_ca_topic_score_gemma":0.0022181778,"teacher_disagreement_score":0.0021939457,"about_ca_system_score_codex":0.00074560574,"about_ca_system_score_gemma":0.0012026861,"threshold_uncertainty_score":0.011602819},"labels":[],"label_agreement":null},{"id":"W2945374007","doi":"10.1007/978-3-030-18305-9_68","title":"Safe Policy Learning with Constrained Return Variance","year":2019,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Computer science; Variance (accounting); Artificial intelligence; Machine learning; Accounting","score_opus":0.011840084781795424,"score_gpt":0.23573608348315536,"score_spread":0.22389599870135993,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2945374007","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0037458206,0.00023904635,0.9924932,0.0001619835,0.00005288815,0.000014664535,0.000027679436,0.0003747368,0.002889927],"genre_scores_gemma":[0.60253125,0.0010166817,0.3637662,0.00044921885,0.0002686558,0.0002706297,0.00033700815,0.0007347132,0.030625748],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9989586,0.00030798707,0.000048373546,0.00020163816,0.00034389913,0.00013947261],"domain_scores_gemma":[0.99689037,0.002287359,0.00015491169,0.0003199109,0.00022220548,0.00012520283],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019145237,0.0011510837,0.0015133375,0.0006215898,0.00045311163,0.001530388,0.001622264,0.0018408622,0.0049590818],"category_scores_gemma":[0.0073868018,0.0008825029,0.00073799695,0.00076108094,0.001824896,0.002052321,0.0029297306,0.0032786452,0.0013099555],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015608527,0.00005924798,0.000247868,0.00010250252,0.000060294115,0.00008898433,0.000072583076,0.69554347,0.0018147525,0.17636545,0.0029400506,0.12254871],"study_design_scores_gemma":[0.00001555234,0.00003000921,0.00003869163,0.00001329855,0.0000070674687,0.000019863877,0.000003879514,0.87992376,0.00071143865,0.11839115,0.00083753205,0.000007830498],"about_ca_topic_score_codex":0.0019225669,"about_ca_topic_score_gemma":0.0013969037,"teacher_disagreement_score":0.0049590818,"about_ca_system_score_codex":0.0011052719,"about_ca_system_score_gemma":0.0017340652,"threshold_uncertainty_score":0.01658976},"labels":[],"label_agreement":null},{"id":"W2945943081","doi":"10.24963/ijcai.2019/581","title":"Metatrace Actor-Critic: Online Step-Size Tuning by Meta-gradient Descent for Reinforcement Learning Control","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Hyperparameter; Reinforcement learning; Computer science; Robustness (evolution); Gradient descent; Artificial intelligence; Nonlinear system; Function approximation; Machine learning; Mathematical optimization; Artificial neural network; Mathematics","score_opus":0.04272323222080878,"score_gpt":0.286499092031481,"score_spread":0.2437758598106722,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2945943081","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00951023,0.00037541732,0.98570025,0.00020709759,0.00006901182,0.000057255438,0.00004064117,0.0017488296,0.002291156],"genre_scores_gemma":[0.73560023,0.00028431063,0.25901893,0.00027157416,0.0000675793,0.00030228496,0.00013866517,0.00048352583,0.003832859],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993507,0.00022996386,0.000035548357,0.00013444602,0.00017503122,0.000074337986],"domain_scores_gemma":[0.99824476,0.00096911134,0.00018362174,0.00023950433,0.0002570587,0.00010592131],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001634012,0.0012847346,0.0013687818,0.00053786125,0.00039748792,0.0011593888,0.0022348026,0.0015425693,0.0032502213],"category_scores_gemma":[0.005835801,0.00065769284,0.00063946564,0.0004783887,0.0011381806,0.0012610868,0.0013532953,0.0024481032,0.00075432693],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009569144,0.000057212488,0.00039054762,0.00006808368,0.000059995007,0.000057363002,0.000038859158,0.9501979,0.0017039502,0.007413169,0.0015689952,0.038348246],"study_design_scores_gemma":[0.000011280851,0.000016416843,0.000019898493,0.0000053419003,0.0000036646618,0.0000073543792,0.0000015622892,0.9972408,0.0003790421,0.002012583,0.00029944547,0.0000026708392],"about_ca_topic_score_codex":0.0032766901,"about_ca_topic_score_gemma":0.0040699504,"teacher_disagreement_score":0.0032766901,"about_ca_system_score_codex":0.0009505824,"about_ca_system_score_gemma":0.0017419984,"threshold_uncertainty_score":0.01087302},"labels":[],"label_agreement":null},{"id":"W2946045694","doi":"10.24963/ijcai.2019/85","title":"A Regularized Opponent Model with Maximum Entropy Objective","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Principle of maximum entropy; Computer science; Inference; Binary number; Mathematical optimization; Probabilistic logic; Iterated function; Random variable; Artificial intelligence; Algorithm; Mathematics","score_opus":0.018995473682165354,"score_gpt":0.23647779733778454,"score_spread":0.21748232365561918,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2946045694","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022733854,0.00015322247,0.9712905,0.0004897422,0.000029303628,0.000046972295,0.00007054081,0.0001915639,0.0049942844],"genre_scores_gemma":[0.7875684,0.00013519931,0.20230894,0.0003310994,0.000049764614,0.00018824938,0.00013775745,0.00011572231,0.009164941],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99923277,0.0003374215,0.000020350537,0.00016328809,0.00015287682,0.000093359085],"domain_scores_gemma":[0.99841607,0.0010488865,0.00017466518,0.000098720564,0.00014925492,0.00011231595],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016417925,0.000696377,0.0011838034,0.0004895339,0.00034134468,0.0010726547,0.0018673597,0.0014774259,0.0038007696],"category_scores_gemma":[0.004295556,0.00046896285,0.0005534468,0.00035269643,0.0015097866,0.0016126849,0.0014464697,0.0015968265,0.00051563565],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008145504,0.0000460375,0.0004966617,0.00006240802,0.00003376808,0.0000818738,0.0000704319,0.88591146,0.0013716669,0.09436473,0.001251274,0.016228132],"study_design_scores_gemma":[0.000009354128,0.000011288712,0.00003214625,0.0000038188327,0.000002273879,0.000008996352,0.0000026497296,0.98737097,0.00011642306,0.012244448,0.0001943367,0.000003253472],"about_ca_topic_score_codex":0.0027511744,"about_ca_topic_score_gemma":0.0025758974,"teacher_disagreement_score":0.0038007696,"about_ca_system_score_codex":0.0011440652,"about_ca_system_score_gemma":0.0012833532,"threshold_uncertainty_score":0.012714863},"labels":[],"label_agreement":null},{"id":"W2946669606","doi":"10.65109/ugvb1408","title":"Building Knowledge for AI Agents with Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Generalization; Artificial intelligence; Function (biology); Reinforcement; Machine learning; Engineering","score_opus":0.01839217257182383,"score_gpt":0.2856183917926542,"score_spread":0.26722621922083034,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2946669606","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008540421,0.0003356022,0.9847483,0.0012976461,0.000059630438,0.00007517634,0.00005539689,0.00047864966,0.004409153],"genre_scores_gemma":[0.4045023,0.00085514266,0.590044,0.00042189553,0.000109224886,0.000384531,0.00018988994,0.0001377842,0.0033551755],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992238,0.00025704617,0.00006976737,0.00013341512,0.00025009876,0.00006574075],"domain_scores_gemma":[0.99795616,0.0010901478,0.00018537858,0.00034722453,0.00028847635,0.00013261709],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016046246,0.0007428235,0.0008214395,0.00054554146,0.00068011053,0.001828856,0.0018821536,0.0016020614,0.002668699],"category_scores_gemma":[0.0075125895,0.00049131754,0.00075950625,0.00043074493,0.0026907518,0.0035050053,0.002677595,0.0029673765,0.00057359255],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006219074,0.00008074973,0.00072104565,0.00018953336,0.00008290537,0.00013208023,0.00027644535,0.6747615,0.0018193502,0.24982679,0.0020812624,0.06996625],"study_design_scores_gemma":[0.000020986905,0.000029256327,0.0000672899,0.000039056657,0.000014834927,0.00002550872,0.000030402072,0.7752737,0.0010001094,0.220061,0.003424834,0.0000130494445],"about_ca_topic_score_codex":0.004386687,"about_ca_topic_score_gemma":0.0050129592,"teacher_disagreement_score":0.004386687,"about_ca_system_score_codex":0.0014514315,"about_ca_system_score_gemma":0.0013880625,"threshold_uncertainty_score":0.010530949},"labels":[],"label_agreement":null},{"id":"W2947183726","doi":"10.1287/moor.2021.1177","title":"On Linear Programming for Constrained and Unconstrained Average-Cost Markov Decision Processes with Countable Action Spaces and Strictly Unbounded Costs","year":2021,"lang":"en","type":"article","venue":"Mathematics of Operations Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Alberta Innovates; Alberta Innovates - Technology Futures; DeepMind; Alberta Machine Intelligence Institute","keywords":"Mathematics; Countable set; Markov decision process; Duality (order theory); Mathematical optimization; State space; Action (physics); Markov kernel; Markov chain; Discrete mathematics; Applied mathematics; Markov process; Markov model; Variable-order Markov model; Statistics","score_opus":0.07719914338480655,"score_gpt":0.38038640943968005,"score_spread":0.3031872660548735,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2947183726","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029828057,0.0011904149,0.95625865,0.0014970915,0.000058853737,0.00007091321,0.00017009652,0.00009345968,0.010832465],"genre_scores_gemma":[0.7909776,0.002291509,0.18928522,0.0007670325,0.00038537086,0.0009511905,0.00051610905,0.00021976311,0.014606229],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9974255,0.0013568581,0.00010198397,0.0004823503,0.0003457097,0.00028742576],"domain_scores_gemma":[0.986129,0.011973132,0.0007867591,0.00021090526,0.00046504635,0.00043514025],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004399921,0.002037339,0.001967076,0.0011824133,0.00081347977,0.0028159355,0.0020074674,0.002406248,0.00479155],"category_scores_gemma":[0.013501312,0.0008966663,0.0016224147,0.0013450615,0.0036703844,0.003977745,0.0033398215,0.0036379858,0.00035106775],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000066629385,0.000100539844,0.00042932772,0.00023123514,0.0000622504,0.0001741293,0.00015602716,0.45309305,0.0006297394,0.5362975,0.0008516857,0.007907815],"study_design_scores_gemma":[0.000016569562,0.000041917654,0.0001230726,0.000035432135,0.000012241385,0.000022259559,0.00003082995,0.76988024,0.00013664376,0.22915216,0.0005339796,0.00001468399],"about_ca_topic_score_codex":0.004662331,"about_ca_topic_score_gemma":0.0036177782,"teacher_disagreement_score":0.00479155,"about_ca_system_score_codex":0.0036955397,"about_ca_system_score_gemma":0.0032649203,"threshold_uncertainty_score":0.02681309},"labels":[],"label_agreement":null},{"id":"W2947766523","doi":"10.24963/ijcai.2019/439","title":"Advantage Amplification in Slowly Evolving Latent-State Environments","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Stylized fact; Abstraction; Computer science; Reinforcement learning; Key (lock); Artificial intelligence; Action (physics); Machine learning; State (computer science); Task (project management); Algorithm; Computer security; Engineering","score_opus":0.018052120014348937,"score_gpt":0.2516236503439548,"score_spread":0.23357153032960584,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2947766523","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.055661105,0.000283885,0.9401986,0.000596614,0.00003420418,0.000052125706,0.00007267501,0.00034331158,0.0027575328],"genre_scores_gemma":[0.91169465,0.00027956627,0.083923556,0.00026601533,0.00006363572,0.00016128292,0.00011008698,0.00013491444,0.0033663702],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974159,0.0011208825,0.00013173674,0.0005076459,0.000517062,0.00030672768],"domain_scores_gemma":[0.9687216,0.026462337,0.0016841844,0.0017233026,0.0007232263,0.00068536145],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004996234,0.0009902196,0.0014328053,0.0007209885,0.00074657163,0.0017583315,0.001766871,0.0014484167,0.003167644],"category_scores_gemma":[0.025944823,0.0008164063,0.0010725663,0.0005260933,0.0022201387,0.0042738533,0.005784857,0.003500861,0.0003540753],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038842525,0.00014884488,0.0032638297,0.00026157915,0.00015878807,0.00048089202,0.0005919006,0.6826917,0.0071090213,0.24692185,0.0017875206,0.05619567],"study_design_scores_gemma":[0.000016931146,0.00005964249,0.00033027754,0.000012844473,0.000013759452,0.00005815635,0.000027516318,0.9084038,0.00089741254,0.08967974,0.00048348564,0.000016392758],"about_ca_topic_score_codex":0.0017146742,"about_ca_topic_score_gemma":0.001815648,"teacher_disagreement_score":0.004996234,"about_ca_system_score_codex":0.0013717839,"about_ca_system_score_gemma":0.001192514,"threshold_uncertainty_score":0.026422977},"labels":[],"label_agreement":null},{"id":"W2948937890","doi":"10.48550/arxiv.1903.03176","title":"MinAtar: An Atari-Inspired Testbed for Thorough and Reproducible Reinforcement Learning Experiments","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":42,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Testbed; Computer science; Representation (politics); Breakout; Artificial intelligence; Set (abstract data type); Human–computer interaction; Toolbox; Machine learning","score_opus":0.11093294896298513,"score_gpt":0.2332999492194836,"score_spread":0.12236700025649848,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2948937890","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3636325,0.0014614178,0.5438235,0.0012675212,0.00095243525,0.0015004973,0.01276325,0.041780904,0.03281798],"genre_scores_gemma":[0.63228744,0.0003919254,0.3489139,0.00044821648,0.000046012738,0.0017142524,0.007849587,0.0024942271,0.005854496],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99907327,0.00038490983,0.00005764649,0.0001871755,0.00018388318,0.000113058544],"domain_scores_gemma":[0.99735343,0.0014848756,0.0001600944,0.0005070961,0.00024815445,0.00024632594],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0020483064,0.0014692422,0.0008491393,0.0004980557,0.0003447204,0.00076517527,0.0036038854,0.0012294474,0.010584092],"category_scores_gemma":[0.0057644425,0.00043831122,0.00067059364,0.0004087429,0.0007107354,0.0011210542,0.0010684385,0.002165598,0.002095865],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003038386,0.0034823741,0.006125166,0.0020900571,0.0005546996,0.0008286452,0.00044827865,0.7718283,0.034480277,0.043516792,0.051739156,0.08186785],"study_design_scores_gemma":[0.0005495095,0.0008443117,0.0015479424,0.00007708123,0.000047669047,0.00011623987,0.00008275628,0.944185,0.016134093,0.016898857,0.01946512,0.00005139224],"about_ca_topic_score_codex":0.0024699252,"about_ca_topic_score_gemma":0.0032008276,"teacher_disagreement_score":0.9979517,"about_ca_system_score_codex":0.00066247897,"about_ca_system_score_gemma":0.0007820145,"threshold_uncertainty_score":0.035407305},"labels":[],"label_agreement":null},{"id":"W2948949475","doi":"10.1109/cybermatics_2018.2018.00148","title":"Optimizing Rescheduling Intervals Through Using Multi-Armed Bandit Algorithms","year":2018,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Athabasca University","funders":"New York State Department of Environmental Conservation","keywords":"Schedule; Computer science; Scheduling (production processes); Job shop scheduling; Reinforcement learning; Mathematical optimization; Operations research; Artificial intelligence; Engineering; Mathematics","score_opus":0.09642049829386522,"score_gpt":0.34390516150211603,"score_spread":0.2474846632082508,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2948949475","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14569522,0.0005831542,0.8491657,0.00037007275,0.000066283734,0.0001666589,0.00005282458,0.0007660981,0.0031339673],"genre_scores_gemma":[0.93927664,0.00012875871,0.05932333,0.000117149284,0.000020920648,0.00013950696,0.00005835906,0.000040358034,0.00089500484],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99881434,0.00050443236,0.00007619926,0.00021099894,0.00020783312,0.00018618087],"domain_scores_gemma":[0.99486744,0.0034415154,0.0008764351,0.0001642563,0.0004280134,0.0002222589],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027560452,0.0012535225,0.0018667527,0.0010625918,0.00067731366,0.0012716741,0.0013493662,0.0014167475,0.0012380324],"category_scores_gemma":[0.007563867,0.0007068501,0.0005839173,0.0005889887,0.0010140665,0.0011531041,0.0007941104,0.0015704564,0.00022299797],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008053044,0.00008568505,0.00052021316,0.000024665884,0.000024678378,0.00002778332,0.000028642602,0.98408705,0.00036758682,0.0013182095,0.00017424261,0.013260768],"study_design_scores_gemma":[0.0000072425455,0.000019827976,0.000040622952,0.0000024498001,0.0000036674357,0.0000024671267,0.000003889457,0.99923563,0.00010575741,0.00054628565,0.000030067851,0.0000020658879],"about_ca_topic_score_codex":0.010001722,"about_ca_topic_score_gemma":0.0068729245,"teacher_disagreement_score":0.010001722,"about_ca_system_score_codex":0.0013550422,"about_ca_system_score_gemma":0.0018101637,"threshold_uncertainty_score":0.01988703},"labels":[],"label_agreement":null},{"id":"W2949600864","doi":"","title":"Model-Based Bayesian Reinforcement Learning in Large Structured Domains","year":2012,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Scalability; Artificial intelligence; Bayesian probability; Machine learning; Bayesian inference","score_opus":0.046170896446990024,"score_gpt":0.20111054857377583,"score_spread":0.15493965212678582,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2949600864","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017977633,0.00030295842,0.9792652,0.00040980792,0.000022453587,0.000029382909,0.000058526693,0.00027210577,0.0016619659],"genre_scores_gemma":[0.84059006,0.00049129094,0.15603657,0.00016534251,0.00006730782,0.00017969715,0.00015627197,0.00008196984,0.0022314547],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990515,0.0004931046,0.000037915586,0.0001622075,0.00016920353,0.000086044965],"domain_scores_gemma":[0.99597025,0.0030200433,0.00031924612,0.00025374503,0.00025116745,0.00018563145],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002037372,0.00086874847,0.0015791218,0.00052593416,0.00046232168,0.001053495,0.0014638562,0.0014577459,0.002246063],"category_scores_gemma":[0.009520848,0.0007454686,0.00057407195,0.0006520157,0.0018996132,0.0025607806,0.0018457148,0.0023667747,0.00028479015],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000049717633,0.000027121721,0.0002799471,0.00005028973,0.000020932035,0.000054096647,0.000050776758,0.9408595,0.00039565444,0.044685107,0.0005336819,0.012993224],"study_design_scores_gemma":[0.00001121406,0.000010285233,0.000035543708,0.0000046158184,0.0000022985532,0.0000057585426,0.0000038108676,0.96464646,0.00007469152,0.035040934,0.00016084455,0.000003503732],"about_ca_topic_score_codex":0.0072382335,"about_ca_topic_score_gemma":0.0065033142,"teacher_disagreement_score":0.0072382335,"about_ca_system_score_codex":0.00153163,"about_ca_system_score_gemma":0.0011827237,"threshold_uncertainty_score":0.014392197},"labels":[],"label_agreement":null},{"id":"W2950048158","doi":"10.48550/arxiv.1404.3328","title":"Myopic Bounds for Optimal Policy of POMDPs: An extension of Lovejoy's structural results","year":2014,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada; Canada Research Chairs","keywords":"Extension (predicate logic); Bounded function; Markov decision process; Mathematical optimization; Relaxation (psychology); Mathematical economics; Upper and lower bounds; Mathematics; Computer science; Markov process; Economics; Statistics","score_opus":0.07299322227582464,"score_gpt":0.23164626624484466,"score_spread":0.15865304396902002,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950048158","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012050999,0.00048648784,0.97539324,0.0010514242,0.000049737875,0.00007154021,0.00019376374,0.00017772165,0.010525106],"genre_scores_gemma":[0.75547004,0.0017021197,0.23483658,0.0007443964,0.00033127665,0.0008673024,0.00051754626,0.00040714745,0.005123606],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99495244,0.0016223117,0.00027025858,0.00086859893,0.0018064214,0.00047997074],"domain_scores_gemma":[0.96498203,0.027492836,0.0023103433,0.0023053088,0.0020466056,0.00086291507],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0069528795,0.0015267004,0.001980463,0.0017330857,0.0011141495,0.002557133,0.0027820214,0.0018026035,0.007739232],"category_scores_gemma":[0.0417317,0.0016906447,0.0018888138,0.0013264808,0.0036213673,0.008055786,0.0047318307,0.006125759,0.0007201696],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000116976975,0.000092814094,0.0005116059,0.00029861447,0.00004642801,0.0000941264,0.0003746765,0.28172326,0.0017788511,0.6936887,0.0018388816,0.01943521],"study_design_scores_gemma":[0.000033937096,0.000075815056,0.00022880392,0.0001091157,0.000015426684,0.000044207733,0.000053611788,0.49878845,0.0011228825,0.49720478,0.002297024,0.000025965446],"about_ca_topic_score_codex":0.0016739912,"about_ca_topic_score_gemma":0.0016688831,"teacher_disagreement_score":0.007739232,"about_ca_system_score_codex":0.0025910472,"about_ca_system_score_gemma":0.0032478692,"threshold_uncertainty_score":0.03677082},"labels":[],"label_agreement":null},{"id":"W2950527138","doi":"10.48550/arxiv.math/0506489","title":"Acceleration Operators in the Value Iteration Algorithms for Markov Decision Processes","year":2005,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Markov decision process; Monotone polygon; Convergence (economics); Mathematical optimization; Algorithm; Markov chain; Operator (biology); Mathematics; Acceleration; Contraction (grammar); Dynamic programming; Computer science; Markov process; Applied mathematics","score_opus":0.06572374678319638,"score_gpt":0.325058715766949,"score_spread":0.25933496898375263,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950527138","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0072198445,0.00022284467,0.9909576,0.00014417851,0.000048329042,0.000032709486,0.0000059359877,0.00006740525,0.0013011674],"genre_scores_gemma":[0.38308442,0.0009864902,0.6100066,0.00021794408,0.00019949829,0.0004167137,0.000051937626,0.00014491308,0.004891465],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9987343,0.00066460785,0.000052515476,0.00014392032,0.00030195044,0.00010279414],"domain_scores_gemma":[0.99547845,0.0036056878,0.00023204707,0.0002202518,0.00032615583,0.00013751035],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035695964,0.0011201239,0.0009920374,0.0005705669,0.0004731687,0.0010058896,0.0012409106,0.0012412671,0.0022705097],"category_scores_gemma":[0.011638335,0.00047527993,0.0009416582,0.0007173872,0.0021565002,0.0020126605,0.0019583046,0.0028020353,0.00047182766],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001407023,0.000112094014,0.0005945765,0.00020767655,0.00005150641,0.00010068062,0.00027399193,0.40674308,0.0039025608,0.5058636,0.0014412339,0.080568336],"study_design_scores_gemma":[0.000017145603,0.00006072033,0.00003577802,0.000016555872,0.0000059537683,0.00002466371,0.000008377171,0.9099413,0.00092065183,0.08793065,0.0010296172,0.000008617537],"about_ca_topic_score_codex":0.0009768584,"about_ca_topic_score_gemma":0.0006511053,"teacher_disagreement_score":0.0035695964,"about_ca_system_score_codex":0.00072168827,"about_ca_system_score_gemma":0.0013167283,"threshold_uncertainty_score":0.018878102},"labels":[],"label_agreement":null},{"id":"W2950564923","doi":"10.48550/arxiv.1906.08649","title":"Exploring Model-based Planning with Policy Networks","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":76,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Benchmarking; Computer science; Code (set theory); Artificial neural network; Sample (material); Control (management); Action (physics); Artificial intelligence; Mathematical optimization; Optimization problem; State space; Machine learning; Algorithm; Mathematics","score_opus":0.19532087655417027,"score_gpt":0.20715320251527736,"score_spread":0.011832325961107087,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950564923","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03235261,0.00029688253,0.9634911,0.00032027907,0.000030973206,0.00004497534,0.00006589174,0.0008455533,0.0025516811],"genre_scores_gemma":[0.7826213,0.00034764045,0.21374032,0.00021199054,0.00003988898,0.00030548597,0.00023489956,0.00019154839,0.0023069002],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995129,0.00020454563,0.000021900398,0.000110735404,0.000095910946,0.00005402178],"domain_scores_gemma":[0.9985368,0.0011404456,0.0000915968,0.00010592876,0.00008105575,0.000044169046],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009928298,0.0011665329,0.0009900328,0.0005337849,0.00038961854,0.0009480074,0.0010623124,0.0011095194,0.002370292],"category_scores_gemma":[0.0038979352,0.0008353175,0.0006733557,0.00046830572,0.001287392,0.001582852,0.0013995257,0.0014926823,0.00033041596],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000026442533,0.000022456792,0.00019478312,0.000028501547,0.000013522434,0.000021688189,0.0000211857,0.98000026,0.00035982873,0.0067033954,0.0002643972,0.012343502],"study_design_scores_gemma":[0.0000053738745,0.0000072650982,0.000012995137,0.0000025167105,0.0000017541862,0.0000024511467,0.0000023216796,0.995417,0.00011385962,0.004285785,0.00014743202,0.0000013103037],"about_ca_topic_score_codex":0.0076205903,"about_ca_topic_score_gemma":0.0069225277,"teacher_disagreement_score":0.0076205903,"about_ca_system_score_codex":0.0012103169,"about_ca_system_score_gemma":0.0016861589,"threshold_uncertainty_score":0.015152514},"labels":[],"label_agreement":null},{"id":"W2950638477","doi":"10.48550/arxiv.1112.1133","title":"Multi-timescale Nexting in a Reinforcement Learning Robot","year":2011,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Alberta Innovates","keywords":"Laptop; Reinforcement learning; Computer science; Feature (linguistics); Range (aeronautics); Artificial intelligence; Robot; Function (biology); Temporal difference learning; Bellman equation; Representation (politics); State (computer science); Dimension (graph theory); Machine learning; Algorithm; Mathematics; Mathematical optimization; Engineering","score_opus":0.1086868290083088,"score_gpt":0.20306236592929322,"score_spread":0.09437553692098442,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950638477","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.38537627,0.00014693412,0.60860795,0.00046888166,0.000068338006,0.00006076636,0.000041640065,0.0018769482,0.0033522104],"genre_scores_gemma":[0.9338747,0.000025658366,0.064573795,0.0000398379,0.000008258105,0.000030182788,0.000020660576,0.000022527034,0.001404418],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998085,0.000049080463,0.000008680336,0.00006235339,0.000042205145,0.000029202716],"domain_scores_gemma":[0.9995252,0.00022632521,0.000052161988,0.000076720484,0.000058208992,0.00006127258],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00065943267,0.00034338952,0.00040305694,0.00015348174,0.0002940163,0.00030586842,0.0008423285,0.0005891015,0.0012785364],"category_scores_gemma":[0.0015423644,0.00023740612,0.00023981236,0.00013426885,0.00071388105,0.000724709,0.0006214938,0.0008502164,0.00020078101],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003879162,0.0002053593,0.0020110172,0.000040463954,0.000026112064,0.00022602733,0.00015604535,0.90531266,0.020622382,0.0062510152,0.0006162528,0.064144775],"study_design_scores_gemma":[0.000019565327,0.000060283604,0.0001728241,0.000001818188,0.0000030843355,0.000017594984,0.0000066253215,0.99574846,0.0018208224,0.0018942786,0.00025023392,0.000004443184],"about_ca_topic_score_codex":0.0036944957,"about_ca_topic_score_gemma":0.002720161,"teacher_disagreement_score":0.0036944957,"about_ca_system_score_codex":0.00046885782,"about_ca_system_score_gemma":0.0004884786,"threshold_uncertainty_score":0.0073459744},"labels":[],"label_agreement":null},{"id":"W2950653703","doi":"10.48550/arxiv.1312.0286","title":"Efficient Learning and Planning with Compressed Predictive States","year":2013,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Reinforcement learning; Artificial intelligence; Observable; A priori and a posteriori; Dimensionality reduction; Machine learning; Curse of dimensionality","score_opus":0.03664535695537425,"score_gpt":0.18255154166165422,"score_spread":0.14590618470627997,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950653703","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010095938,0.00014396735,0.9872875,0.00027489822,0.000021111045,0.000037582497,0.00010423576,0.0004737969,0.0015608653],"genre_scores_gemma":[0.7004469,0.00037331894,0.29539198,0.00021008044,0.000074427546,0.00029586023,0.0005593043,0.00017078912,0.0024774026],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999328,0.0002376546,0.000035494024,0.00014627463,0.0001854379,0.00006708978],"domain_scores_gemma":[0.99721646,0.0020052502,0.000245294,0.00028951166,0.00017126225,0.000072264185],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010098591,0.00090553105,0.000979194,0.0005481923,0.00036722425,0.00093915605,0.001289345,0.0011032788,0.0023283872],"category_scores_gemma":[0.0065574367,0.00070875156,0.0006140697,0.0006632474,0.0015564286,0.0022165806,0.0017655116,0.002158036,0.00029573118],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005951306,0.00002427054,0.00020232868,0.000051872285,0.000015235988,0.000050028397,0.0000679678,0.9387379,0.00076115463,0.035745498,0.000799688,0.023484576],"study_design_scores_gemma":[0.000005394124,0.000008023008,0.000021122454,0.0000046379837,0.0000017599738,0.0000057071898,0.00000460978,0.98242193,0.00025472167,0.017074138,0.00019522366,0.0000026983214],"about_ca_topic_score_codex":0.0063104345,"about_ca_topic_score_gemma":0.0064442297,"teacher_disagreement_score":0.0063104345,"about_ca_system_score_codex":0.0011302459,"about_ca_system_score_gemma":0.001528716,"threshold_uncertainty_score":0.012547433},"labels":[],"label_agreement":null},{"id":"W2950989964","doi":"","title":"Apprenticeship Learning using Inverse Reinforcement Learning and Gradient Methods","year":2012,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":73,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Markov decision process; Computer science; Function (biology); Artificial intelligence; Mathematical optimization; Inverse; Apprenticeship; Gradient method; Machine learning; Algorithm; Markov process; Mathematics; Statistics","score_opus":0.13644769773097773,"score_gpt":0.2590478520903364,"score_spread":0.12260015435935864,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950989964","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009749476,0.00021417736,0.9881891,0.000094995135,0.000028143457,0.000041877003,0.0000071113304,0.00023033824,0.0014448854],"genre_scores_gemma":[0.6465748,0.00031617418,0.34811914,0.00015953429,0.00007012788,0.00022981099,0.000060270377,0.00010389226,0.0043663047],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99936,0.00023467685,0.000034946657,0.00014682005,0.00016320987,0.000060399176],"domain_scores_gemma":[0.9985563,0.0008594098,0.00014648584,0.00012593737,0.00022240453,0.00008942177],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014188247,0.0010677731,0.0012568388,0.00059631176,0.00031669525,0.0007622296,0.0016092254,0.0012638374,0.0019727526],"category_scores_gemma":[0.004439262,0.0005386878,0.0005127872,0.0003153037,0.0012572638,0.001285434,0.0013431688,0.0015041219,0.00047961602],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008001884,0.00013987484,0.0009395681,0.000116772426,0.000062869345,0.000098803095,0.00011182071,0.86015207,0.0021339813,0.02660183,0.00092754676,0.1086348],"study_design_scores_gemma":[0.000010855503,0.000034637473,0.000055028726,0.0000053013478,0.0000033833435,0.000015395342,0.000003671233,0.9935675,0.0004297769,0.0055123065,0.0003575398,0.000004682744],"about_ca_topic_score_codex":0.0022400294,"about_ca_topic_score_gemma":0.0015585567,"teacher_disagreement_score":0.0022400294,"about_ca_system_score_codex":0.0005795319,"about_ca_system_score_gemma":0.0009312978,"threshold_uncertainty_score":0.007503569},"labels":[],"label_agreement":null},{"id":"W2951110424","doi":"10.48550/arxiv.1612.02088","title":"Transition-based versus State-based Reward Functions for MDPs with Value-at-Risk","year":2016,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Markov decision process; Reinforcement learning; Function (biology); Action (physics); Transformation (genetics); Bellman equation; Markov process; Markov chain; State (computer science); Current (fluid); Order (exchange); Computer science; Mathematical optimization; Econometrics; Mathematics; Economics; Artificial intelligence; Statistics; Machine learning; Engineering; Algorithm; Finance","score_opus":0.05494755650732404,"score_gpt":0.1875184610339302,"score_spread":0.13257090452660614,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2951110424","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012189625,0.0005293076,0.98589337,0.00029036045,0.000029424953,0.000033296765,0.000027196458,0.0000870407,0.00092047616],"genre_scores_gemma":[0.8162323,0.0012031184,0.1784309,0.00021224863,0.000103228165,0.0002741453,0.00017247103,0.00015553666,0.003216005],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9965107,0.0021340773,0.00017738243,0.0004974443,0.00042645592,0.00025400694],"domain_scores_gemma":[0.9816621,0.015933285,0.0009149249,0.00041426584,0.0007292716,0.000346173],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0066805626,0.0016041511,0.0019173194,0.0008628182,0.00046247573,0.0016840374,0.0012854469,0.0017321843,0.001981599],"category_scores_gemma":[0.022874009,0.00066718255,0.0011332207,0.0007567951,0.0024811493,0.0037747442,0.0020495914,0.0027967163,0.00028368935],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000090767506,0.00004977293,0.0005259269,0.00011272987,0.000041217747,0.00004800802,0.00007865761,0.9013988,0.0004063885,0.08068662,0.00027592055,0.016285202],"study_design_scores_gemma":[0.000006297461,0.000027621185,0.00006911678,0.000012545645,0.0000074329078,0.000008999156,0.000006978277,0.9720861,0.00015750232,0.027468959,0.00014043578,0.000008101454],"about_ca_topic_score_codex":0.002866604,"about_ca_topic_score_gemma":0.0015111695,"teacher_disagreement_score":0.0066805626,"about_ca_system_score_codex":0.0023971754,"about_ca_system_score_gemma":0.0018781851,"threshold_uncertainty_score":0.035330594},"labels":[],"label_agreement":null},{"id":"W2951553872","doi":"","title":"SOLAR: Deep Structured Representations for Model-Based Reinforcement Learning","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Robotics; Range (aeronautics); Simple (philosophy); Linear-quadratic regulator; Offline learning; Quadratic equation; Image (mathematics); Machine learning; Control (management); Robot; Online learning; Mathematics; Engineering","score_opus":0.0705683000237207,"score_gpt":0.2240496447580726,"score_spread":0.15348134473435188,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2951553872","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0035302176,0.00014430928,0.9926267,0.00017403834,0.000057659036,0.000029351442,0.00015236865,0.0016514362,0.0016338222],"genre_scores_gemma":[0.55438155,0.0004309752,0.43520287,0.00042870847,0.00011181983,0.00049079815,0.001281187,0.0006783358,0.00699383],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99968493,0.00010033735,0.000016965838,0.000066321845,0.00010150044,0.000030010024],"domain_scores_gemma":[0.99922633,0.00038688743,0.00007645486,0.00015153736,0.00011127226,0.00004745484],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008328533,0.0009263203,0.000787622,0.00039854538,0.00027473818,0.00093918026,0.0015067015,0.0011980868,0.0055061923],"category_scores_gemma":[0.0038902375,0.00053410174,0.0006359236,0.0004592341,0.0006919852,0.0014988212,0.001448567,0.0024922534,0.001193912],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000084927175,0.00007317487,0.0003192085,0.00009561157,0.00004671598,0.00005239472,0.000040953306,0.85650605,0.002224689,0.04022343,0.006793636,0.093539245],"study_design_scores_gemma":[0.0000061452147,0.000009616067,0.000014875356,0.0000039718857,0.000001928383,0.0000038820826,0.0000013497585,0.98805386,0.00030012603,0.011023125,0.00057910004,0.0000019947604],"about_ca_topic_score_codex":0.003021113,"about_ca_topic_score_gemma":0.003948487,"teacher_disagreement_score":0.0055061923,"about_ca_system_score_codex":0.0008322253,"about_ca_system_score_gemma":0.0009986935,"threshold_uncertainty_score":0.01842004},"labels":[],"label_agreement":null},{"id":"W2952014622","doi":"10.48550/arxiv.1503.04269","title":"An Emphatic Approach to the Problem of Off-policy Temporal-Difference Learning","year":2015,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Temporal difference learning; Discounting; Lambda; Bootstrapping (finance); Computer science; Computation; Parametric statistics; Function (biology); Algorithm; State (computer science); Applied mathematics; Artificial intelligence; Mathematics; Reinforcement learning; Statistics; Econometrics","score_opus":0.09377041807705082,"score_gpt":0.21911309879659377,"score_spread":0.12534268071954296,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2952014622","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0025921715,0.00016852838,0.994605,0.000822049,0.000086052474,0.000020998381,0.000016685406,0.000110400775,0.0015780298],"genre_scores_gemma":[0.52150285,0.00053373625,0.468243,0.0013586133,0.00047097832,0.0002253387,0.00008718183,0.00022973974,0.0073485035],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9978962,0.0009707258,0.00009491129,0.00047712107,0.00046406809,0.000096941025],"domain_scores_gemma":[0.9926998,0.005169953,0.000383419,0.0010256947,0.0005246325,0.00019659371],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005564186,0.0008741574,0.0011346067,0.00040008174,0.00060632214,0.0015833051,0.0033389155,0.0025862076,0.0038828533],"category_scores_gemma":[0.024854142,0.0007724694,0.0005578867,0.00055787957,0.0032478864,0.00385627,0.0028659906,0.006619625,0.0005176471],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037540574,0.00016161769,0.000929482,0.00030382822,0.00010357553,0.00015000922,0.00027429173,0.37590358,0.0043274243,0.4715724,0.0043603494,0.1415381],"study_design_scores_gemma":[0.000029834138,0.00008490399,0.000092618124,0.000023998045,0.000011749811,0.000069828675,0.000014203584,0.87655485,0.0014665844,0.11867785,0.0029582172,0.000015392752],"about_ca_topic_score_codex":0.0016160519,"about_ca_topic_score_gemma":0.0010688799,"teacher_disagreement_score":0.005564186,"about_ca_system_score_codex":0.0012824783,"about_ca_system_score_gemma":0.0015459962,"threshold_uncertainty_score":0.029426575},"labels":[],"label_agreement":null},{"id":"W2952348496","doi":"10.3389/fnbot.2018.00032","title":"Experience Replay Using Transition Sequences","year":2018,"lang":"en","type":"article","venue":"Frontiers in Neurorobotics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Singapore University of Technology and Design; Ministry of Education - Singapore; Ministry of Education, India; University of Alberta","keywords":"Reinforcement learning; Computer science; Construct (python library); Artificial intelligence; State space; Space (punctuation); Transition (genetics); State (computer science); Function (biology); Machine learning; Algorithm","score_opus":0.027803115942547044,"score_gpt":0.27424405348465974,"score_spread":0.24644093754211271,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2952348496","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07451227,0.0003986396,0.9213085,0.00019403044,0.00006841452,0.00015349347,0.000080877515,0.0014800222,0.0018038024],"genre_scores_gemma":[0.8387589,0.00017985236,0.15856348,0.00011425931,0.000027231912,0.00023586604,0.00021117866,0.00012220732,0.0017870037],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99882394,0.00045264335,0.000084477484,0.00028503675,0.0002608179,0.00009314696],"domain_scores_gemma":[0.99478865,0.002592608,0.00062841753,0.0011826324,0.00057322136,0.00023447785],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017807573,0.0011392549,0.0008978405,0.00063745305,0.00048454019,0.00084324356,0.0012111022,0.0010362742,0.0024612644],"category_scores_gemma":[0.014613226,0.0005397965,0.0005819072,0.00039708152,0.0010107331,0.0022107342,0.0015255444,0.0015534164,0.00061371835],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012142928,0.0003508926,0.0054558143,0.00030942637,0.00015496768,0.0004277726,0.00069243205,0.642372,0.022700999,0.020614272,0.0020231006,0.30368406],"study_design_scores_gemma":[0.00007622706,0.00052866194,0.00095858163,0.000048091348,0.000037666385,0.00020861001,0.00008430897,0.9637264,0.013175246,0.017692486,0.0034198358,0.00004390006],"about_ca_topic_score_codex":0.0017342806,"about_ca_topic_score_gemma":0.0016725662,"teacher_disagreement_score":0.0024612644,"about_ca_system_score_codex":0.00054983096,"about_ca_system_score_gemma":0.00072147476,"threshold_uncertainty_score":0.009417713},"labels":[],"label_agreement":null},{"id":"W2952662670","doi":"10.48550/arxiv.1906.04328","title":"Importance Resampling for Off-policy Prediction","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Resampling; Consistency (knowledge bases); Variance (accounting); Computer science; Reinforcement learning; Sampling (signal processing); Jackknife resampling; Function (biology); Statistics; Artificial intelligence; Machine learning; Econometrics; Mathematics","score_opus":0.08829860924234678,"score_gpt":0.21524033782485585,"score_spread":0.12694172858250907,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2952662670","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034211528,0.0002466288,0.9626913,0.00025116114,0.00007084683,0.00011998217,0.000042011965,0.000659048,0.0017075267],"genre_scores_gemma":[0.856603,0.00015029777,0.14067982,0.00026424005,0.00007191511,0.00021691251,0.00014686906,0.00016517217,0.0017018426],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976603,0.0010318675,0.00011254089,0.00041710044,0.00058044074,0.0001977067],"domain_scores_gemma":[0.9889714,0.0074648582,0.0007912026,0.0014247658,0.0010539625,0.00029381973],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004552013,0.0010228943,0.0013423343,0.00071315723,0.00051341543,0.0011079473,0.0017231725,0.0010755572,0.0021736585],"category_scores_gemma":[0.026537128,0.0006236212,0.000543117,0.00046402676,0.0012652098,0.0019437658,0.0013166481,0.0021668382,0.00037024415],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045759545,0.00030186758,0.0041495482,0.00013768705,0.00011241452,0.00016991497,0.00016263653,0.81932664,0.005652366,0.030911382,0.0024533004,0.13616459],"study_design_scores_gemma":[0.000016845705,0.00005826627,0.00022196553,0.000009142627,0.0000076019605,0.000015689991,0.0000089953755,0.9899677,0.0013928391,0.007926964,0.00036706313,0.0000069190232],"about_ca_topic_score_codex":0.0038919675,"about_ca_topic_score_gemma":0.004134072,"teacher_disagreement_score":0.004552013,"about_ca_system_score_codex":0.001072415,"about_ca_system_score_gemma":0.0013652724,"threshold_uncertainty_score":0.0240736},"labels":[],"label_agreement":null},{"id":"W2952735266","doi":"10.48550/arxiv.1205.2606","title":"Exploring compact reinforcement-learning representations with linear regression","year":2012,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Reinforcement; Regression; Linear regression; Artificial intelligence; Computer science; Machine learning; Mathematics; Statistics; Psychology; Social psychology","score_opus":0.21910190354494213,"score_gpt":0.2371817989078626,"score_spread":0.018079895362920456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2952735266","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00905223,0.000112055335,0.9896096,0.00011533824,0.0000095502055,0.000015824284,0.000018420747,0.0004377005,0.0006292155],"genre_scores_gemma":[0.60444057,0.00026551387,0.3912122,0.00023032699,0.000045307384,0.00025141286,0.00022727114,0.00031463144,0.0030127796],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992161,0.0003471863,0.000035561436,0.00018308088,0.00014310815,0.00007494659],"domain_scores_gemma":[0.9973494,0.002028085,0.00019157058,0.00021918834,0.0001532591,0.00005852789],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015619945,0.00095616636,0.0014469044,0.00049149786,0.00031191175,0.0010690688,0.0014565255,0.0012181352,0.0024573607],"category_scores_gemma":[0.008461122,0.00068383495,0.00063536287,0.00065183674,0.0013182529,0.002700787,0.0020336553,0.0021481232,0.0006763703],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000080294,0.000066308625,0.00040167564,0.00007910563,0.00003028283,0.00004942695,0.00009736128,0.89392585,0.0011766518,0.04417369,0.0010556207,0.058863837],"study_design_scores_gemma":[0.000008351461,0.000010861881,0.000013211491,0.000003603144,0.0000018364418,0.0000056836116,0.0000047863955,0.9850347,0.0001855173,0.01455934,0.0001696733,0.000002403794],"about_ca_topic_score_codex":0.0023702788,"about_ca_topic_score_gemma":0.0017281323,"teacher_disagreement_score":0.0024573607,"about_ca_system_score_codex":0.0009280638,"about_ca_system_score_gemma":0.00093380216,"threshold_uncertainty_score":0.008260727},"labels":[],"label_agreement":null},{"id":"W2953243993","doi":"10.48550/arxiv.1906.08226","title":"Unsupervised State Representation Learning in Atari","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke; Université de Montréal","funders":"","keywords":"Representation (politics); Computer science; Benchmark (surveying); Feature learning; Artificial intelligence; Generative grammar; Variety (cybernetics); Generative model; Encoder; Code (set theory); Machine learning; State (computer science); Encoding (memory)","score_opus":0.0804774768729731,"score_gpt":0.20423136063171912,"score_spread":0.12375388375874602,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2953243993","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016502563,0.00022267934,0.97553444,0.00048092546,0.00005710981,0.000115639596,0.00034322272,0.0017406167,0.00500287],"genre_scores_gemma":[0.5463634,0.00022600972,0.4419813,0.00043257224,0.000081680235,0.0006020314,0.001697192,0.00071085786,0.007904981],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99861014,0.00057187706,0.000054757413,0.00041251653,0.00024779863,0.00010280027],"domain_scores_gemma":[0.9975241,0.0012225019,0.00022169437,0.00066139526,0.00024236306,0.00012803476],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018269584,0.00096731546,0.001234579,0.00065161404,0.0006642738,0.0012506102,0.00332825,0.0017784804,0.005461422],"category_scores_gemma":[0.008599085,0.00050060684,0.0010223673,0.00071956555,0.0016042764,0.0030527478,0.0027612217,0.003681345,0.0012521048],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019419318,0.00023137276,0.0011574178,0.00022346318,0.000100495155,0.00013323226,0.00025709983,0.6699487,0.0041702003,0.17188844,0.008370222,0.14332515],"study_design_scores_gemma":[0.000015820517,0.000030183892,0.0001013784,0.000009990234,0.0000059779354,0.000019626803,0.000010801811,0.94580823,0.0008687044,0.05197534,0.0011441212,0.0000097096645],"about_ca_topic_score_codex":0.0050103744,"about_ca_topic_score_gemma":0.0071274326,"teacher_disagreement_score":0.005461422,"about_ca_system_score_codex":0.0016641112,"about_ca_system_score_gemma":0.0017091851,"threshold_uncertainty_score":0.018270314},"labels":[],"label_agreement":null},{"id":"W2953351167","doi":"10.48550/arxiv.1704.02544","title":"A Linearly Relaxed Approximate Linear Program for Markov Decision Processes","year":2017,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Linear programming; Constraint (computer-aided design); Markov decision process; Mathematical optimization; Markov chain; Markov process; Computer science; Mathematics; Linear approximation; Applied mathematics; Nonlinear system; Statistics; Physics; Machine learning","score_opus":0.08128687683702827,"score_gpt":0.24218287870035018,"score_spread":0.16089600186332192,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2953351167","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009666622,0.00037696882,0.98477477,0.00042396857,0.00003223372,0.00008286419,0.00020926798,0.00020670137,0.004226614],"genre_scores_gemma":[0.4679749,0.00094148505,0.5205617,0.00047811551,0.00018700308,0.00092325534,0.0009779789,0.0002496548,0.007705903],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99752134,0.0011396281,0.000084828556,0.0004928998,0.00053296564,0.00022837667],"domain_scores_gemma":[0.9955733,0.0034214947,0.0003308957,0.00019874287,0.0003037996,0.0001717072],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020475765,0.0013550177,0.0014009819,0.00057938276,0.00036359526,0.0017835551,0.001439574,0.0014968081,0.004581667],"category_scores_gemma":[0.009473356,0.00064406026,0.001034341,0.0009986616,0.0014626336,0.001990258,0.0022677602,0.0037086585,0.0006688314],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011718058,0.00009934721,0.00035857427,0.00018610095,0.000038823015,0.00010110031,0.00010602418,0.87178373,0.00128037,0.09571069,0.0025013227,0.027716856],"study_design_scores_gemma":[0.000009626922,0.000030362404,0.000034139357,0.000013302508,0.0000047598996,0.000017532506,0.000009876878,0.97425824,0.00022678051,0.024705252,0.0006849413,0.000005183044],"about_ca_topic_score_codex":0.0027371608,"about_ca_topic_score_gemma":0.002290647,"teacher_disagreement_score":0.004581667,"about_ca_system_score_codex":0.0014615825,"about_ca_system_score_gemma":0.001873793,"threshold_uncertainty_score":0.015327156},"labels":[],"label_agreement":null},{"id":"W2954388198","doi":"10.1109/icit.2019.8755084","title":"Evaluating Architecture Impacts on Deep Imitation Learning Performance for Autonomous Driving","year":2019,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Computer science; Architecture; Imitation; Deep learning; Artificial intelligence; Human–computer interaction; Computer architecture; Psychology; Geography","score_opus":0.02746936062123937,"score_gpt":0.3030474467729648,"score_spread":0.2755780861517254,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2954388198","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96636194,0.00076850905,0.030190421,0.00015627575,0.000048368132,0.000042378288,0.0000850305,0.0003504399,0.00199658],"genre_scores_gemma":[0.99682546,0.00007926788,0.0027579062,0.000008391096,0.0000029690307,0.0000124774215,0.00005949512,0.00000990986,0.00024414642],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997998,0.00004222652,0.000017362385,0.0000418634,0.000038265007,0.000060536266],"domain_scores_gemma":[0.99829406,0.000977711,0.00016092137,0.00014016428,0.0003215744,0.000105618274],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010064006,0.0007232557,0.00035806798,0.0003741318,0.00020859491,0.00037782677,0.00044347602,0.0007924043,0.0008557421],"category_scores_gemma":[0.004554518,0.00024855242,0.00026877096,0.00019006705,0.0003332314,0.0006759704,0.0005524895,0.0006629545,0.00017329775],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034675965,0.0001414515,0.006203567,0.00009495898,0.00005365694,0.000042572618,0.000033858116,0.9533602,0.004870327,0.0004677255,0.0002776221,0.03410729],"study_design_scores_gemma":[0.000008094971,0.00026241408,0.001521413,0.000007974913,0.000019689316,0.00001322226,0.0000129398495,0.99328834,0.004521599,0.000256538,0.000081589635,0.0000062526847],"about_ca_topic_score_codex":0.0054401523,"about_ca_topic_score_gemma":0.0044382527,"teacher_disagreement_score":0.0054401523,"about_ca_system_score_codex":0.0005409644,"about_ca_system_score_gemma":0.00044180852,"threshold_uncertainty_score":0.010816932},"labels":[],"label_agreement":null},{"id":"W2955319481","doi":"10.1002/aic.16689","title":"Toward self‐driving processes: A deep reinforcement learning approach to control","year":2019,"lang":"en","type":"article","venue":"AIChE Journal","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":144,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Self driving; Reinforcement; Self-control; Control (management); Computer science; Artificial intelligence; Engineering; Psychology; Social psychology; Transport engineering","score_opus":0.014074329873777026,"score_gpt":0.23225183349437672,"score_spread":0.2181775036205997,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2955319481","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023878245,0.00020008105,0.9720256,0.00030636272,0.000028526982,0.00002496858,0.000013860498,0.0001966041,0.0033256775],"genre_scores_gemma":[0.9386797,0.00015818771,0.05854211,0.00011069623,0.000032699503,0.00006692056,0.000019265513,0.000027896855,0.0023624892],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998074,0.000066864166,0.000008103092,0.000033698427,0.000053802392,0.000030185474],"domain_scores_gemma":[0.99947625,0.0002724335,0.00006976323,0.000040040035,0.000108742985,0.000032719265],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000731565,0.0004913674,0.00050179235,0.00025200238,0.00024581648,0.00054311287,0.0008532794,0.00073129707,0.0013471647],"category_scores_gemma":[0.0015220345,0.00025331185,0.00031922915,0.00020479785,0.0009241971,0.00056112674,0.0007718704,0.0012389637,0.00013103304],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000011976403,0.000018601017,0.000185559,0.000015117142,0.0000110746505,0.000016492653,0.000018741312,0.9805617,0.0008845752,0.008834914,0.00018973308,0.00925148],"study_design_scores_gemma":[0.0000017129905,0.000005428817,0.00001531612,9.659152e-7,8.7881347e-7,0.0000012040997,6.9473276e-7,0.9983663,0.00007373533,0.0014631327,0.000069908936,7.906e-7],"about_ca_topic_score_codex":0.005890329,"about_ca_topic_score_gemma":0.0040549715,"teacher_disagreement_score":0.005890329,"about_ca_system_score_codex":0.0008092059,"about_ca_system_score_gemma":0.00074854033,"threshold_uncertainty_score":0.011712134},"labels":[],"label_agreement":null},{"id":"W2956872166","doi":"10.48550/arxiv.1907.04651","title":"Incrementally Learning Functions of the Return","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Taylor series; Value (mathematics); Computer science; Temporal difference learning; Mathematics; Mathematical optimization; Applied mathematics; Artificial intelligence; Machine learning; Mathematical analysis","score_opus":0.05402498703515711,"score_gpt":0.17527060647856768,"score_spread":0.12124561944341057,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2956872166","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019000288,0.00010708592,0.97967786,0.00012837253,0.000024670455,0.00002213648,0.000033221608,0.00028437076,0.0007220343],"genre_scores_gemma":[0.77056956,0.00023702884,0.22548072,0.00012848685,0.000052460942,0.00016844882,0.00013564886,0.00012429239,0.0031033745],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99953187,0.00013496529,0.000027475144,0.00012373565,0.00013167571,0.000050297156],"domain_scores_gemma":[0.9963238,0.0025266942,0.00030741654,0.00027574963,0.0004514531,0.00011487843],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014475335,0.00061994646,0.00078978844,0.00060931774,0.00018738853,0.0009421733,0.0012705293,0.0009755024,0.0020385445],"category_scores_gemma":[0.012538748,0.00041760886,0.00048058163,0.0004312893,0.0008556413,0.0018114428,0.0009419982,0.0015745905,0.00047801834],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023618319,0.00012726033,0.0020446992,0.00010784181,0.000052002048,0.00008047876,0.00009478927,0.77009815,0.009733146,0.04372067,0.001780925,0.17192386],"study_design_scores_gemma":[0.000005798071,0.00001826989,0.00010409626,0.000004610589,0.0000040232526,0.000011656787,0.0000025761317,0.99184155,0.00094234943,0.0068723224,0.00018788484,0.0000048457177],"about_ca_topic_score_codex":0.0019051834,"about_ca_topic_score_gemma":0.0020654073,"teacher_disagreement_score":0.0020385445,"about_ca_system_score_codex":0.0010028663,"about_ca_system_score_gemma":0.00095849636,"threshold_uncertainty_score":0.007655382},"labels":[],"label_agreement":null},{"id":"W2962845991","doi":"10.1609/aaai.v32i1.11775","title":"OptionGAN: Learning Joint Reward-Policy Options Using Generative Adversarial Inverse Reinforcement Learning","year":2018,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Open Philanthropy Project; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Leverage (statistics); Computer science; Adversarial system; Artificial intelligence; Generative grammar; Machine learning; Function (biology); Set (abstract data type)","score_opus":0.0495135636832109,"score_gpt":0.29439513433302306,"score_spread":0.24488157064981214,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2962845991","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015594696,0.00024829974,0.9807891,0.00025136562,0.00004164995,0.000056217377,0.0000731185,0.000866705,0.002078901],"genre_scores_gemma":[0.82524884,0.00023595286,0.16799425,0.00047700616,0.000054705743,0.00032270452,0.00031971204,0.0002323251,0.0051145316],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995301,0.00018125732,0.000017755876,0.000098184566,0.00011777531,0.000054994674],"domain_scores_gemma":[0.99870265,0.0009483231,0.000086520056,0.00010898708,0.00008629427,0.00006719374],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013422219,0.0010928556,0.0010207488,0.00042591462,0.00024324236,0.0006623615,0.0013785898,0.0015107874,0.003138666],"category_scores_gemma":[0.004129024,0.00064372947,0.0006564279,0.00031900097,0.0013779222,0.0013235089,0.0016879058,0.0021539489,0.00051738264],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007112381,0.000037814403,0.0004983076,0.00004713445,0.000043149754,0.0000779366,0.000036886227,0.9540747,0.0011994997,0.01308177,0.0011656441,0.02966597],"study_design_scores_gemma":[0.0000066988946,0.000014152269,0.000037259353,0.000005296806,0.0000028430063,0.000011631162,0.0000019158385,0.99311215,0.00029472564,0.0062954244,0.00021339057,0.0000045699835],"about_ca_topic_score_codex":0.0024566173,"about_ca_topic_score_gemma":0.002901818,"teacher_disagreement_score":0.003138666,"about_ca_system_score_codex":0.0007007261,"about_ca_system_score_gemma":0.0010145091,"threshold_uncertainty_score":0.010499895},"labels":[],"label_agreement":null},{"id":"W2962985403","doi":"","title":"Universal Successor Representations for Transfer Reinforcement Learning","year":2018,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Initialization; Successor cardinal; Artificial intelligence; Set (abstract data type); Transfer of learning; Focus (optics); Function (biology); Machine learning; Reinforcement; Value (mathematics); Mathematics; Programming language","score_opus":0.057285571163360936,"score_gpt":0.34720038236061634,"score_spread":0.28991481119725543,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2962985403","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024776345,0.0003484821,0.9707762,0.00027115596,0.000056698136,0.000048753664,0.000065386674,0.00074205,0.0029149747],"genre_scores_gemma":[0.8784943,0.00029742578,0.11664903,0.00019692627,0.00003879576,0.0002539477,0.00018516454,0.00010830671,0.003776105],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99950695,0.00019190062,0.000029338373,0.000117438016,0.00009533032,0.000059132923],"domain_scores_gemma":[0.9985262,0.0007918054,0.00014902448,0.0002620974,0.00017771001,0.00009309008],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015603653,0.0007799943,0.0009175434,0.0004774175,0.00029276538,0.0008244071,0.0011304531,0.001103663,0.0036814706],"category_scores_gemma":[0.006114143,0.00033014014,0.0005086372,0.00038234328,0.0011982545,0.0021629331,0.0013924284,0.001996082,0.0005133658],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013433432,0.00012717342,0.00073440856,0.00015519475,0.000055817334,0.00012464708,0.00019967122,0.75987417,0.0026822116,0.113068335,0.0021882732,0.12065583],"study_design_scores_gemma":[0.000012171304,0.00004215223,0.00005505778,0.000012627292,0.000006018766,0.000017950833,0.00000837243,0.95071477,0.00049500354,0.048018727,0.0006098914,0.000007313164],"about_ca_topic_score_codex":0.0013193475,"about_ca_topic_score_gemma":0.001420355,"teacher_disagreement_score":0.0036814706,"about_ca_system_score_codex":0.00095114636,"about_ca_system_score_gemma":0.0009395495,"threshold_uncertainty_score":0.01231581},"labels":[],"label_agreement":null},{"id":"W2963097726","doi":"10.1609/aaai.v33i01.33014384","title":"The Utility of Sparse Representations for Control in Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Sparse approximation; Computer science; Neural coding; Bootstrapping (finance); Artificial intelligence; Representation (politics); Machine learning; Artificial neural network; Locality; Mathematics","score_opus":0.06591031971543576,"score_gpt":0.3089931647799222,"score_spread":0.24308284506448646,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963097726","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013887132,0.00020694279,0.9828636,0.0005426632,0.00003160598,0.000035034212,0.000029035356,0.00009669245,0.002307247],"genre_scores_gemma":[0.8551536,0.00049197447,0.14118366,0.00038403284,0.00012891821,0.00019405829,0.00006512964,0.0000676764,0.002330898],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989793,0.000507112,0.000046051722,0.00016518074,0.00021211893,0.00009024705],"domain_scores_gemma":[0.9930421,0.005407297,0.0004770858,0.0005513339,0.00034296935,0.00017921028],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025749907,0.0007978194,0.0009166682,0.0004942686,0.00042845833,0.0011940502,0.0010785131,0.0012988268,0.0021725781],"category_scores_gemma":[0.01558663,0.00041238024,0.00063486956,0.00049094885,0.0025542304,0.0027879395,0.0014939401,0.0027479657,0.00027064708],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010732662,0.00008588642,0.00070964306,0.00012713714,0.0000389486,0.00008694277,0.00015420582,0.6488802,0.0025968386,0.3052908,0.00090151204,0.041020643],"study_design_scores_gemma":[0.000014177014,0.000063716274,0.000053588312,0.000011187979,0.00000550539,0.000013118336,0.000007717714,0.91054195,0.00049549347,0.08841015,0.00037584428,0.000007458019],"about_ca_topic_score_codex":0.0017873199,"about_ca_topic_score_gemma":0.001221239,"teacher_disagreement_score":0.0025749907,"about_ca_system_score_codex":0.00094523444,"about_ca_system_score_gemma":0.0008674151,"threshold_uncertainty_score":0.013618052},"labels":[],"label_agreement":null},{"id":"W2963142324","doi":"10.1609/aaai.v32i1.11831","title":"When Waiting Is Not an Option: Learning Options With a Deliberation Cost","year":2018,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":80,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada; Institut de Valorisation des Données","keywords":"Deliberation; Interpretability; Bounded rationality; Computer science; Rationality; Work (physics); Management science; Risk analysis (engineering); Artificial intelligence; Epistemology; Economics; Business; Political science; Engineering; Philosophy","score_opus":0.03604037314581124,"score_gpt":0.27441794202672404,"score_spread":0.2383775688809128,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963142324","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15659806,0.00031107775,0.8356711,0.0014165286,0.00006111731,0.00008321868,0.00009438936,0.00061066414,0.005153897],"genre_scores_gemma":[0.8918637,0.00015228658,0.10530346,0.00019764395,0.000027848044,0.000104080646,0.000097732394,0.000087516404,0.0021656984],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985519,0.0006350838,0.000079419326,0.00035564398,0.00020828501,0.00016969374],"domain_scores_gemma":[0.98929006,0.008310358,0.0007840408,0.0006753033,0.0003963495,0.0005438957],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028983708,0.00074195047,0.0009069122,0.00032375116,0.0004744066,0.0011752544,0.00148567,0.00161757,0.002877244],"category_scores_gemma":[0.019386325,0.00048421702,0.00058398914,0.000343495,0.0022693584,0.004393625,0.0023332404,0.0028126824,0.00033067373],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012118875,0.00022639838,0.0038693969,0.0002120657,0.00012654498,0.00040650653,0.0005409782,0.7377026,0.0032684377,0.14413565,0.0016935422,0.10660601],"study_design_scores_gemma":[0.00006918348,0.000096686534,0.00029051417,0.000023732433,0.000023259383,0.00004908965,0.00005354647,0.8649562,0.0013327233,0.13246651,0.00061602023,0.000022540067],"about_ca_topic_score_codex":0.0023260822,"about_ca_topic_score_gemma":0.002416314,"teacher_disagreement_score":0.0028983708,"about_ca_system_score_codex":0.00096200715,"about_ca_system_score_gemma":0.0015535074,"threshold_uncertainty_score":0.0153282285},"labels":[],"label_agreement":null},{"id":"W2963250930","doi":"","title":"Convergent Tree Backup and Retrace with Function Approximation","year":2018,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Université de Montréal","funders":"","keywords":"Backup; Reinforcement learning; Saddle point; Computer science; Tree (set theory); Function approximation; Mathematical optimization; Convergence (economics); Function (biology); Approximation algorithm; Bootstrapping (finance); Saddle; Quadratic equation; Applied mathematics; Mathematics; Algorithm; Artificial intelligence; Combinatorics","score_opus":0.0275852956681716,"score_gpt":0.265515831660857,"score_spread":0.2379305359926854,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963250930","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014164275,0.0002851525,0.98220414,0.00036187287,0.000064690554,0.00006951856,0.000043103966,0.00046762082,0.0023396243],"genre_scores_gemma":[0.664749,0.00024543074,0.3271852,0.00043101018,0.00007358113,0.00033274773,0.00024053612,0.0003394242,0.006403112],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99897766,0.00034070946,0.00004797934,0.00020931511,0.00030036434,0.00012398836],"domain_scores_gemma":[0.99499714,0.0031335226,0.00033326139,0.0006610859,0.0005613175,0.00031371572],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027103839,0.001239705,0.0020847884,0.0006847101,0.0007663841,0.0011593611,0.0024604544,0.0020925356,0.0039389734],"category_scores_gemma":[0.013823671,0.0006008911,0.00081363984,0.0006384702,0.0017738305,0.0019754244,0.0028198923,0.0034232798,0.0009254823],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021781611,0.000114996554,0.0008011697,0.00013682537,0.000042069503,0.00014111106,0.00013747005,0.8479833,0.0015891506,0.052664753,0.0033454245,0.09282597],"study_design_scores_gemma":[0.000015264513,0.00003464485,0.00003866708,0.000009706613,0.0000030756823,0.000023725104,0.000007728343,0.9866701,0.00051751896,0.012279235,0.0003959235,0.000004388615],"about_ca_topic_score_codex":0.003130248,"about_ca_topic_score_gemma":0.0026095058,"teacher_disagreement_score":0.0039389734,"about_ca_system_score_codex":0.0012483745,"about_ca_system_score_gemma":0.0018247485,"threshold_uncertainty_score":0.014334083},"labels":[],"label_agreement":null},{"id":"W2963274839","doi":"","title":"Reinforcement Learning based Embodied Agents Modelling Human Users Through Interaction and Multi-Sensory Perception","year":2017,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Task (project management); Reinforcement learning; Embodied cognition; Perception; Artificial intelligence; Control (management); Human–computer interaction; Complement (music); Feedback control; Machine learning; Control engineering; Engineering","score_opus":0.20392061309279874,"score_gpt":0.2604560799500727,"score_spread":0.05653546685727395,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963274839","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13027261,0.00013518355,0.86476743,0.00031089445,0.000029528057,0.000058160804,0.000046704467,0.0003389576,0.0040405975],"genre_scores_gemma":[0.95156276,0.00007525283,0.045350697,0.00003591219,0.000009525318,0.000069239926,0.00002048527,0.000022056349,0.0028539882],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996117,0.0001876194,0.00001738284,0.00008132808,0.00005914989,0.000042931624],"domain_scores_gemma":[0.99835217,0.0010192556,0.0002418491,0.00016286313,0.00012704471,0.000096932425],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00077053806,0.00060805946,0.00048672446,0.0002300735,0.00023390891,0.00084765337,0.00093545445,0.0008406322,0.0023585027],"category_scores_gemma":[0.0037989537,0.00038126044,0.00045960658,0.00018342814,0.0010789196,0.0010605893,0.0010285582,0.0009673785,0.00030697475],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009033621,0.00005975695,0.0010563252,0.00003649335,0.000031418756,0.000090243295,0.00024287553,0.97431695,0.0031603046,0.009667816,0.00013148981,0.011115921],"study_design_scores_gemma":[0.0000074231175,0.00003243336,0.00011712573,0.0000027137246,0.000003932377,0.000010473785,0.00001080153,0.9963102,0.0003347693,0.0029909944,0.00017437724,0.0000047398335],"about_ca_topic_score_codex":0.0038907807,"about_ca_topic_score_gemma":0.0030065046,"teacher_disagreement_score":0.0038907807,"about_ca_system_score_codex":0.00064537855,"about_ca_system_score_gemma":0.00058676815,"threshold_uncertainty_score":0.007889926},"labels":[],"label_agreement":null},{"id":"W2963313316","doi":"","title":"Scalable trust-region method for deep reinforcement learning using Kronecker-factored approximation","year":2017,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":92,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Kronecker delta; Reinforcement learning; Scalability; Computer science; Trust region; Curvature; Mathematical optimization; Artificial intelligence; Theoretical computer science; Mathematics","score_opus":0.05248858218001681,"score_gpt":0.31582601486795703,"score_spread":0.2633374326879402,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963313316","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032099248,0.00011480442,0.9954531,0.00006285336,0.000024418177,0.00001922329,0.000016422851,0.00047168805,0.0006275103],"genre_scores_gemma":[0.6120471,0.00025445668,0.38086823,0.00021224408,0.00006415586,0.00024049134,0.00020840677,0.0005440684,0.0055609574],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994174,0.00019350075,0.00003163057,0.00012670738,0.0001654074,0.000065430715],"domain_scores_gemma":[0.9985531,0.00073899195,0.0001284989,0.00016467186,0.000319976,0.0000948447],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016288992,0.0011775487,0.00143103,0.00047561122,0.00031625477,0.000889598,0.001461519,0.0013071438,0.003410135],"category_scores_gemma":[0.0047955196,0.0005657082,0.00076601387,0.00039276318,0.0011254905,0.0013617987,0.0012217264,0.0022321418,0.0009275754],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009053696,0.000033267035,0.00034060018,0.000053527416,0.000036987298,0.000054888464,0.000044479206,0.9456026,0.0021222527,0.012071946,0.0013589445,0.038189936],"study_design_scores_gemma":[0.0000035015617,0.000009075788,0.000011748515,0.0000019621796,0.0000015400899,0.0000041542116,0.000001163436,0.99808323,0.00023399065,0.0014884241,0.00015937735,0.0000018515744],"about_ca_topic_score_codex":0.006236748,"about_ca_topic_score_gemma":0.004733259,"teacher_disagreement_score":0.006236748,"about_ca_system_score_codex":0.0015075713,"about_ca_system_score_gemma":0.0017275171,"threshold_uncertainty_score":0.012400925},"labels":[],"label_agreement":null},{"id":"W2963388106","doi":"10.1609/aaai.v33i01.33015797","title":"QUOTA: The Quantile Option Architecture for Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Quantile; Reinforcement learning; Architecture; Computer science; Optimism; Value (mathematics); Artificial intelligence; Econometrics; Machine learning; Economics; Psychology","score_opus":0.05244733210608453,"score_gpt":0.2880631182674069,"score_spread":0.23561578616132234,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963388106","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012469952,0.00033118442,0.9828524,0.00035371174,0.00007661223,0.000050364233,0.00004914277,0.0006704391,0.0031461485],"genre_scores_gemma":[0.8616918,0.00034334694,0.13284807,0.0003169359,0.000072982824,0.00023864512,0.000108736225,0.00012958638,0.0042500133],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999337,0.00027986683,0.000034265333,0.000118438555,0.00015476909,0.000075640215],"domain_scores_gemma":[0.99891686,0.0005653277,0.000085142696,0.0001286924,0.00019389283,0.00011008432],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016729366,0.000687212,0.0009121712,0.0003510092,0.0004076825,0.00112451,0.0016724118,0.0009906929,0.0065372856],"category_scores_gemma":[0.004551687,0.00036206105,0.00046466946,0.00042295197,0.0011653906,0.0018061883,0.0016494934,0.002009379,0.00077079434],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029257053,0.00013231846,0.0012948766,0.00012474868,0.00008845281,0.00009812408,0.00012702367,0.7556628,0.0030874792,0.1163745,0.0025941334,0.120122984],"study_design_scores_gemma":[0.000020749974,0.000046549,0.000074698015,0.000008611348,0.0000073571046,0.0000141383825,0.0000055282285,0.96073836,0.00039740873,0.03771995,0.0009589583,0.000007670334],"about_ca_topic_score_codex":0.0022950189,"about_ca_topic_score_gemma":0.0021415611,"teacher_disagreement_score":0.0065372856,"about_ca_system_score_codex":0.0008151458,"about_ca_system_score_gemma":0.0010724963,"threshold_uncertainty_score":0.021869421},"labels":[],"label_agreement":null},{"id":"W2963430540","doi":"10.1609/aaai.v33i01.33015789","title":"ACE: An Actor Ensemble Algorithm for Continuous Control with Tree Search","year":2019,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Ensemble learning; Search tree; Computer science; Perspective (graphical); Tree (set theory); Control (management); Artificial intelligence; Value (mathematics); Machine learning; Search algorithm; Ensemble forecasting; Mathematical optimization; Algorithm; Mathematics","score_opus":0.048752940312597276,"score_gpt":0.28943699690266234,"score_spread":0.24068405659006506,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963430540","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0038089545,0.00018096258,0.9943263,0.00006667293,0.000033910866,0.000025463205,0.000020741318,0.00034202478,0.0011950716],"genre_scores_gemma":[0.47477204,0.00031216137,0.5196019,0.0002777235,0.00010563435,0.0003652769,0.00021573753,0.00026287214,0.0040867003],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99952626,0.00015205788,0.000023824903,0.00009821248,0.00014190734,0.000057767036],"domain_scores_gemma":[0.9989931,0.0006457747,0.0000709913,0.00007511209,0.00015276861,0.000062237035],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013085029,0.00090444024,0.0014423103,0.00059364195,0.0004169865,0.00070431264,0.0013557199,0.0012894007,0.0032222376],"category_scores_gemma":[0.0027414763,0.00045742447,0.0006004969,0.00054090435,0.0006718455,0.0010751786,0.0012676688,0.0017606263,0.00067035534],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007624582,0.00003499321,0.00036467027,0.00004287212,0.000056592606,0.000037148508,0.00004302588,0.9001227,0.00095939013,0.012317141,0.0014459217,0.08449926],"study_design_scores_gemma":[0.000005287057,0.000014155861,0.000018232024,0.0000031331258,0.0000028850798,0.0000051818442,0.0000017169422,0.9976222,0.00011455291,0.0018905558,0.0003198601,0.000002192943],"about_ca_topic_score_codex":0.0038300082,"about_ca_topic_score_gemma":0.0036386964,"teacher_disagreement_score":0.0038300082,"about_ca_system_score_codex":0.0005468339,"about_ca_system_score_gemma":0.000998823,"threshold_uncertainty_score":0.01077944},"labels":[],"label_agreement":null},{"id":"W2963437270","doi":"","title":"Per-decision Multi-step Temporal Difference Learning with Control Variates.","year":2018,"lang":"en","type":"article","venue":"Uncertainty in Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; University of British Columbia","funders":"","keywords":"Control variates; Temporal difference learning; Reinforcement learning; Variance (accounting); Computer science; Control (management); Variance reduction; Monte Carlo method; Artificial intelligence; Machine learning; Statistics; Mathematics; Hybrid Monte Carlo; Bayesian probability; Markov chain Monte Carlo","score_opus":0.04175902341515242,"score_gpt":0.29736944076871763,"score_spread":0.2556104173535652,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963437270","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0150369,0.0004085802,0.9818672,0.00020042792,0.0001219127,0.000084040345,0.000035256482,0.0004402206,0.001805525],"genre_scores_gemma":[0.81434983,0.00014282008,0.18253028,0.0002726714,0.00006013165,0.00019093913,0.000115177114,0.00007669919,0.002261473],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990398,0.00035441297,0.00006389794,0.0001917184,0.0002497193,0.00010030368],"domain_scores_gemma":[0.9945697,0.0039750487,0.000328714,0.00041619022,0.00044056945,0.0002696925],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031541125,0.0008672708,0.001187087,0.0003521859,0.00038948722,0.0010013717,0.0020299812,0.0015201783,0.0038520577],"category_scores_gemma":[0.011967219,0.00042734883,0.0005111022,0.00043423395,0.0012225646,0.0016814348,0.0016631442,0.0023946385,0.0005230919],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005693145,0.00029013449,0.0013716241,0.00016541188,0.000083884006,0.00008796369,0.00008131689,0.848229,0.0017650869,0.021626385,0.0013684622,0.12436139],"study_design_scores_gemma":[0.00001291764,0.000041629988,0.000041364994,0.000004391926,0.0000039319248,0.000009207711,0.0000019637112,0.9969494,0.0004561492,0.00225368,0.00022153326,0.0000038650674],"about_ca_topic_score_codex":0.0028337608,"about_ca_topic_score_gemma":0.002731928,"teacher_disagreement_score":0.0038520577,"about_ca_system_score_codex":0.0009492963,"about_ca_system_score_gemma":0.0016424592,"threshold_uncertainty_score":0.016680777},"labels":[],"label_agreement":null},{"id":"W2963462732","doi":"","title":"Planning by Prioritized Sweeping with Small Backups","year":2013,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Backup; Successor cardinal; Computer science; Reinforcement learning; Flexibility (engineering); Computation; Implementation; Process (computing); Distributed computing; State (computer science); Artificial intelligence; Algorithm; Mathematics","score_opus":0.03698794599683398,"score_gpt":0.27144154936796033,"score_spread":0.23445360337112636,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963462732","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.060151707,0.00022972662,0.9341944,0.00021389505,0.00008121847,0.000097708136,0.00010494677,0.0023107608,0.0026156995],"genre_scores_gemma":[0.7448787,0.00012413015,0.25252843,0.00009963955,0.000027022852,0.00017651102,0.00013595715,0.00019356977,0.0018360124],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994961,0.00011430168,0.000040622308,0.00012730023,0.00013873292,0.00008301013],"domain_scores_gemma":[0.9982339,0.0008953833,0.00014272021,0.0004440965,0.00013419718,0.00014969894],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00093853177,0.0007804367,0.0008157133,0.00042471127,0.00048157902,0.0007641969,0.0014583437,0.00065919635,0.005164178],"category_scores_gemma":[0.0036685716,0.0004869588,0.00044983227,0.00038382417,0.00095077255,0.0014300043,0.0015053486,0.0015403691,0.00056518544],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009631538,0.00023366498,0.0014723531,0.0001891261,0.00006277036,0.00029409246,0.0002748831,0.7062021,0.020839116,0.026359016,0.0025149784,0.24059477],"study_design_scores_gemma":[0.00006826109,0.00008937803,0.00012900372,0.000011245612,0.00001343824,0.00004721026,0.000025383642,0.9798544,0.0037283085,0.014811961,0.0012098495,0.00001154878],"about_ca_topic_score_codex":0.0027193527,"about_ca_topic_score_gemma":0.0030980115,"teacher_disagreement_score":0.005164178,"about_ca_system_score_codex":0.00046719576,"about_ca_system_score_gemma":0.0012939163,"threshold_uncertainty_score":0.01727587},"labels":[],"label_agreement":null},{"id":"W2963558674","doi":"","title":"Context-dependent upper-confidence bounds for directed exploration","year":2018,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Context (archaeology); Computation; Machine learning; Artificial intelligence; Variance (accounting); Mathematical optimization; Algorithm; Mathematics","score_opus":0.03561658697522566,"score_gpt":0.2804357901936462,"score_spread":0.24481920321842052,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963558674","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005013588,0.0003254371,0.9927763,0.00016702057,0.0000241778,0.00003245762,0.000042972468,0.00025088733,0.0013672059],"genre_scores_gemma":[0.65687937,0.00084246125,0.3375142,0.00050046877,0.00013551119,0.0006586001,0.00038616583,0.0004584333,0.0026247157],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.996012,0.0011050999,0.00031077777,0.00078370346,0.0014375359,0.00035093902],"domain_scores_gemma":[0.9560038,0.03540703,0.0023419121,0.0022983253,0.0031108102,0.0008383035],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005353631,0.0019450828,0.0017906645,0.0018009278,0.00069926854,0.0025135549,0.003203173,0.0022895497,0.004003085],"category_scores_gemma":[0.06969464,0.0011267536,0.0013855703,0.0010714454,0.0027817183,0.0046640467,0.0046916073,0.006025323,0.0008666658],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001018265,0.00007040889,0.00096989295,0.00013680696,0.000048218433,0.00005295486,0.00011099693,0.9151457,0.001422248,0.05186247,0.00088037015,0.029198145],"study_design_scores_gemma":[0.000007767883,0.00002369187,0.00007964202,0.000029414372,0.000005969854,0.000013717674,0.0000051390534,0.9804952,0.00067889126,0.018412378,0.00023847638,0.000009772924],"about_ca_topic_score_codex":0.0030350073,"about_ca_topic_score_gemma":0.0031813718,"teacher_disagreement_score":0.005353631,"about_ca_system_score_codex":0.0022000605,"about_ca_system_score_gemma":0.0025544916,"threshold_uncertainty_score":0.02831304},"labels":[],"label_agreement":null},{"id":"W2963800416","doi":"","title":"Policy Error Bounds for Model-Based Reinforcement Learning with Factored Linear Models","year":2016,"lang":"en","type":"article","venue":"Conference on Learning Theory","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Mathematical proof; Markov decision process; Contraction (grammar); Property (philosophy); Computer science; Mathematics; Applied mathematics; Mathematical optimization; Measure (data warehouse); Class (philosophy); Markov process; Linear model; Artificial intelligence; Machine learning; Statistics","score_opus":0.0519210604033747,"score_gpt":0.29348081808446674,"score_spread":0.24155975768109206,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963800416","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0063606007,0.00045502812,0.99134314,0.00025506367,0.000033442913,0.000026450321,0.000032357275,0.00018847651,0.0013054783],"genre_scores_gemma":[0.7997343,0.0012611941,0.19450915,0.00040110797,0.00018374555,0.0003439819,0.00028641848,0.0004472861,0.0028328835],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.994065,0.0023949104,0.00025518608,0.0008922321,0.0018695086,0.0005233228],"domain_scores_gemma":[0.9556433,0.03736877,0.0023474721,0.0016904434,0.0021744694,0.00077564345],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010712652,0.0024552154,0.002485827,0.0014519452,0.00070729444,0.0027375028,0.0024957566,0.0020374262,0.0035137935],"category_scores_gemma":[0.053588845,0.001047204,0.0015622132,0.0009388541,0.0035172044,0.005631635,0.0041454956,0.0054751392,0.0005184544],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000082671955,0.000044768967,0.0003499541,0.00011850876,0.000046120982,0.000031256102,0.00006980357,0.92758083,0.0005475566,0.061563723,0.00033969342,0.00922502],"study_design_scores_gemma":[0.000003741375,0.000026393956,0.000024617033,0.00001769848,0.0000055682135,0.000007766376,0.0000049034124,0.9742284,0.00024304901,0.025304802,0.00012738045,0.0000056586678],"about_ca_topic_score_codex":0.0036762585,"about_ca_topic_score_gemma":0.0020528946,"teacher_disagreement_score":0.010712652,"about_ca_system_score_codex":0.004202228,"about_ca_system_score_gemma":0.0033205461,"threshold_uncertainty_score":0.056654632},"labels":[],"label_agreement":null},{"id":"W2963828709","doi":"10.48550/arxiv.1703.01327","title":"Multi-step Reinforcement Learning: A Unifying Algorithm","year":2017,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Backup; Algorithm; TRACE (psycholinguistics); Sampling (signal processing); Focus (optics); Monte Carlo method; Importance sampling; Artificial intelligence; Machine learning; Mathematics","score_opus":0.09250465055229885,"score_gpt":0.2190304302502291,"score_spread":0.12652577969793027,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963828709","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0048288894,0.0001189792,0.9931251,0.00018629778,0.000026746118,0.000058505644,0.000010612118,0.0004922294,0.0011526808],"genre_scores_gemma":[0.3091453,0.00019517483,0.68678457,0.00035332527,0.00007673487,0.00033374797,0.00007956664,0.00022208925,0.0028094442],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979808,0.00059608085,0.00013353972,0.00057449867,0.00053473277,0.00018048323],"domain_scores_gemma":[0.9966968,0.002049168,0.00020932671,0.0005162494,0.00034288547,0.00018547954],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045319973,0.0011883056,0.0016746449,0.00083966955,0.00060975784,0.0013558833,0.0033071658,0.002346471,0.0024171292],"category_scores_gemma":[0.010071719,0.0006054158,0.00083049043,0.0006262771,0.0017704463,0.0027736402,0.00305521,0.0031281507,0.0006646661],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021241455,0.00018600278,0.0013223301,0.000114770606,0.00006961561,0.00007407265,0.00020717029,0.6845416,0.0025508057,0.09311544,0.0018161738,0.2157896],"study_design_scores_gemma":[0.000020679536,0.000041663483,0.00002935244,0.000009154924,0.000005421021,0.00001368784,0.0000042149836,0.9863766,0.00049496646,0.012500602,0.0004977522,0.0000059078197],"about_ca_topic_score_codex":0.0022895748,"about_ca_topic_score_gemma":0.002148053,"teacher_disagreement_score":0.0045319973,"about_ca_system_score_codex":0.0013481246,"about_ca_system_score_gemma":0.002275206,"threshold_uncertainty_score":0.023967743},"labels":[],"label_agreement":null},{"id":"W2963923407","doi":"","title":"Addressing Function Approximation Error in Actor-Critic Methods","year":2018,"lang":"en","type":"article","venue":"UvA-DARE (University of Amsterdam)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":414,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Bellman equation; Suite; Function (biology); Function approximation; Value (mathematics); Limit (mathematics); Artificial intelligence; Approximation algorithm; Temporal difference learning; Mathematical optimization; Machine learning; Algorithm; Artificial neural network; Mathematics","score_opus":0.07700182874909071,"score_gpt":0.32152854316397317,"score_spread":0.24452671441488244,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963923407","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009835275,0.00047612583,0.98743546,0.000329293,0.00007493994,0.000032859414,0.000016041135,0.00043270836,0.0013673914],"genre_scores_gemma":[0.80200016,0.00036415388,0.19258495,0.00042756085,0.00011010162,0.00020759214,0.00007741661,0.0002744173,0.0039535426],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979382,0.0009318911,0.00011555525,0.0003452617,0.0004675464,0.00020152981],"domain_scores_gemma":[0.9891884,0.007904612,0.00064609986,0.00065880595,0.0013162337,0.00028597814],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006675546,0.0020486065,0.0023057384,0.00061333564,0.0006164487,0.001655963,0.002410543,0.0025844134,0.0022060573],"category_scores_gemma":[0.021899395,0.0009211313,0.00049267826,0.0005594149,0.002085866,0.0020395091,0.002371957,0.0037562521,0.0006170324],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016423837,0.000057959558,0.0009682485,0.0001208867,0.00007296595,0.00008064903,0.00009485982,0.93587637,0.0012595796,0.015829923,0.0013392206,0.044135112],"study_design_scores_gemma":[0.000013650466,0.000023256369,0.000038993658,0.000011860635,0.0000060123857,0.000012575158,0.0000051130696,0.99263966,0.00048797682,0.0065069096,0.0002494765,0.0000045615234],"about_ca_topic_score_codex":0.0033529368,"about_ca_topic_score_gemma":0.0026676394,"teacher_disagreement_score":0.006675546,"about_ca_system_score_codex":0.0013608608,"about_ca_system_score_gemma":0.001935544,"threshold_uncertainty_score":0.03530407},"labels":[],"label_agreement":null},{"id":"W2963969179","doi":"10.3389/fnbot.2019.00052","title":"A Novel Model for Arbitration Between Planning and Habitual Control Systems","year":2019,"lang":"en","type":"article","venue":"PubMed Central","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Action selection; Reinforcement learning; Internal model; Task (project management); Control (management); Artificial intelligence; Action (physics); A priori and a posteriori; Kinematics; Machine learning; Human–computer interaction","score_opus":0.030970041851913634,"score_gpt":0.2354315636919563,"score_spread":0.20446152184004268,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963969179","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035379175,0.00022502276,0.95332825,0.0004274745,0.00008523048,0.00004969033,0.00013986058,0.0011311839,0.009234159],"genre_scores_gemma":[0.9380776,0.00016671955,0.050198317,0.000118989665,0.000038764138,0.00016249136,0.00011304545,0.00006585612,0.011058136],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997458,0.0000433375,0.00001250269,0.00009390397,0.00005336733,0.00005100706],"domain_scores_gemma":[0.9996928,0.00010286286,0.000060156824,0.000050087772,0.000057560686,0.000036589438],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000405551,0.00053549843,0.0005458887,0.00019635595,0.0002982586,0.0009016757,0.0014451853,0.0008859974,0.0044676824],"category_scores_gemma":[0.00091813056,0.00032414246,0.0005072177,0.00020462842,0.00089996075,0.0011081024,0.00089108857,0.0015346477,0.00058188254],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013093758,0.00007945303,0.0008580788,0.00007722113,0.000055662214,0.00019791281,0.00011435171,0.8833047,0.008844712,0.0742712,0.0016418169,0.030423973],"study_design_scores_gemma":[0.000012799642,0.000027463562,0.00009335826,0.0000028059865,0.000005871615,0.000023758055,0.0000024192966,0.98925203,0.0005241205,0.009152945,0.00089779194,0.0000046475593],"about_ca_topic_score_codex":0.003195533,"about_ca_topic_score_gemma":0.0028373192,"teacher_disagreement_score":0.0044676824,"about_ca_system_score_codex":0.000722626,"about_ca_system_score_gemma":0.0009705163,"threshold_uncertainty_score":0.014945924},"labels":[],"label_agreement":null},{"id":"W2964112145","doi":"","title":"Weak convergence properties of constrained emphatic temporal-difference learning with constant and slowly diminishing stepsize","year":2016,"lang":"en","type":"article","venue":"Journal of Machine Learning Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Iterated function; Convergence (economics); Markov decision process; Markov chain; Mathematics; Weak convergence; Constant (computer programming); Ergodic theory; Applied mathematics; Divergence (linguistics); Rate of convergence; Limit (mathematics); Stochastic approximation; Markov process; Mathematical optimization; Computer science; Key (lock); Mathematical analysis; Statistics","score_opus":0.05478746984954716,"score_gpt":0.3014922333272734,"score_spread":0.2467047634777262,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964112145","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032825395,0.00027494863,0.9640053,0.000345799,0.000037880054,0.00005851728,0.00005684409,0.00011310534,0.0022822341],"genre_scores_gemma":[0.7757737,0.00054939365,0.21744907,0.00045276177,0.000077840814,0.0004724268,0.00026536174,0.00017048467,0.0047889403],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9982431,0.0005964383,0.00012785206,0.0004022361,0.00047981367,0.00015064726],"domain_scores_gemma":[0.98508215,0.010890375,0.0009459664,0.00077797304,0.0017790209,0.00052440213],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0059929797,0.0011278898,0.0013852662,0.001114993,0.0005666818,0.0015305505,0.0020281891,0.0018186684,0.0028015797],"category_scores_gemma":[0.038796928,0.00056163693,0.0010949089,0.00065078,0.0030318177,0.0030020415,0.0027448575,0.0026916144,0.0004023402],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021771023,0.0001017542,0.0026691833,0.0002440151,0.00009709453,0.00018202518,0.00021978174,0.72633827,0.003662004,0.24487782,0.00069169234,0.020698627],"study_design_scores_gemma":[0.000011119582,0.0000346737,0.0000865622,0.000013988802,0.000004547766,0.000017875269,0.0000075875405,0.9717881,0.00047269676,0.027373685,0.00018018397,0.000008965323],"about_ca_topic_score_codex":0.0029481032,"about_ca_topic_score_gemma":0.001475567,"teacher_disagreement_score":0.0059929797,"about_ca_system_score_codex":0.0015212955,"about_ca_system_score_gemma":0.0016951673,"threshold_uncertainty_score":0.031694293},"labels":[],"label_agreement":null},{"id":"W2964190622","doi":"","title":"Eigenoption Discovery through the Deep Successor Representation","year":2017,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Successor cardinal; Representation (politics); Feature learning; Exploit; Deep learning; Leverage (statistics); ENCODE; Machine learning; Intuition; Cognitive science; Mathematics","score_opus":0.09020491085728016,"score_gpt":0.22714582750763676,"score_spread":0.1369409166503566,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964190622","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.060402457,0.00024311434,0.93486667,0.0005499081,0.000032561547,0.000051243947,0.00007901413,0.0004158595,0.003359196],"genre_scores_gemma":[0.8243931,0.0001336637,0.17177397,0.00017573836,0.00001758075,0.00013491906,0.00010612706,0.00009134828,0.0031735145],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996081,0.00016033623,0.000020946127,0.00008859988,0.00006500851,0.000056895195],"domain_scores_gemma":[0.99852306,0.0009887225,0.00015408818,0.00014437607,0.000091005546,0.0000987835],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013154367,0.00055866176,0.0008535112,0.0005763117,0.0005076707,0.0012273449,0.0010906097,0.0012626234,0.0025549985],"category_scores_gemma":[0.0048608435,0.0004704713,0.00062341325,0.0004184543,0.0016515984,0.0025908009,0.0018305524,0.0018069958,0.00031646347],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021938886,0.00010896713,0.0020848413,0.00010710295,0.000051395124,0.00014850682,0.00025300944,0.6871555,0.0037468818,0.18691066,0.002468025,0.11674577],"study_design_scores_gemma":[0.0000128923175,0.000028418619,0.00007654836,0.00001293081,0.000004747326,0.000021490034,0.000013840964,0.93801993,0.0006695645,0.060740802,0.00039032847,0.000008376192],"about_ca_topic_score_codex":0.0012393974,"about_ca_topic_score_gemma":0.0018421954,"teacher_disagreement_score":0.0025549985,"about_ca_system_score_codex":0.00087391457,"about_ca_system_score_gemma":0.0010306711,"threshold_uncertainty_score":0.008547366},"labels":[],"label_agreement":null},{"id":"W2964191931","doi":"","title":"Unsupervised Video Object Segmentation for Deep Reinforcement Learning","year":2018,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Segmentation; Representation (politics); Object (grammar); Motion (physics); Code (set theory); Unsupervised learning; Object detection; Focus (optics); Computer vision; Action (physics); Suite; Feature learning; Deep learning; Machine learning; Set (abstract data type)","score_opus":0.0505535799427427,"score_gpt":0.20287374622500598,"score_spread":0.15232016628226328,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964191931","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0056260936,0.00014630571,0.9915643,0.00013901322,0.000029350296,0.000035225832,0.000057598198,0.0009699704,0.0014321263],"genre_scores_gemma":[0.57098883,0.0002685237,0.42243135,0.0002545559,0.00006384393,0.00031667354,0.00039154265,0.00032173682,0.00496288],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99961275,0.00010565204,0.000017620869,0.00011900911,0.00009695398,0.000047975383],"domain_scores_gemma":[0.9990675,0.000510788,0.00012797413,0.00012114266,0.00011501695,0.000057516507],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010260165,0.0008856543,0.0007984735,0.0004890698,0.0003119605,0.0006768045,0.0014168527,0.0010011056,0.004240522],"category_scores_gemma":[0.0038267476,0.0004892582,0.0004999338,0.0004340155,0.0009752417,0.0011551342,0.00097363506,0.0017691895,0.0006532142],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010381075,0.0000803748,0.000589411,0.00008298709,0.000042914333,0.00006175187,0.000044935423,0.87321454,0.0039620176,0.02903226,0.0024697508,0.09031525],"study_design_scores_gemma":[0.000005298687,0.000009687232,0.00003486225,0.0000039179467,0.0000019820263,0.0000047688022,0.000001389281,0.9910782,0.0005477325,0.007882025,0.00042769127,0.0000023850005],"about_ca_topic_score_codex":0.0057500997,"about_ca_topic_score_gemma":0.006056781,"teacher_disagreement_score":0.0057500997,"about_ca_system_score_codex":0.0018273905,"about_ca_system_score_gemma":0.0013301268,"threshold_uncertainty_score":0.014185905},"labels":[],"label_agreement":null},{"id":"W2964227312","doi":"10.1609/aaai.v31i1.10916","title":"The Option-Critic Architecture","year":2017,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":695,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Flexibility (engineering); Computer science; Abstraction; Architecture; Key (lock); Artificial intelligence; Machine learning; Computer security; Economics; Management; Epistemology","score_opus":0.014118730041843058,"score_gpt":0.26830135844916925,"score_spread":0.2541826284073262,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964227312","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008548323,0.00036373406,0.97952914,0.00083041965,0.000082678496,0.000040682924,0.00007408186,0.0005093313,0.010021665],"genre_scores_gemma":[0.69831157,0.00075820286,0.28758433,0.00041881908,0.00012676202,0.00027963534,0.00016300892,0.00014691602,0.012210756],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994716,0.00015200289,0.000029531726,0.00013550602,0.00016076383,0.00005060763],"domain_scores_gemma":[0.9991235,0.00041659648,0.00007339577,0.00015151512,0.00014128597,0.00009372817],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00095617137,0.00080011366,0.0006322312,0.00034410474,0.00045258348,0.0011863173,0.0015195655,0.0013400256,0.0040588384],"category_scores_gemma":[0.0037740686,0.000599154,0.0005988979,0.0003453439,0.0017240167,0.0024004735,0.0017975662,0.0032127972,0.00085941685],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008669641,0.00003656457,0.000623595,0.00010435884,0.00007058808,0.00017058688,0.00011504713,0.5908361,0.0034288524,0.34608105,0.002975099,0.05547143],"study_design_scores_gemma":[0.000017980634,0.000023219834,0.00009602516,0.000016364465,0.0000148894915,0.00004263932,0.0000057689863,0.86341226,0.0005979657,0.1329641,0.002794942,0.000013800475],"about_ca_topic_score_codex":0.002566592,"about_ca_topic_score_gemma":0.003196265,"teacher_disagreement_score":0.0040588384,"about_ca_system_score_codex":0.00087842153,"about_ca_system_score_gemma":0.0013681393,"threshold_uncertainty_score":0.0135781765},"labels":[],"label_agreement":null},{"id":"W2964295739","doi":"","title":"Imagination-Augmented Agents for Deep Reinforcement Learning","year":2017,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":163,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Robustness (evolution); Computer science; Artificial intelligence; Construct (python library); Policy learning; Machine learning; Contrast (vision); Deep learning; Context (archaeology); Architecture","score_opus":0.0722518834787979,"score_gpt":0.22074985081393533,"score_spread":0.14849796733513743,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964295739","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0057955435,0.0003030326,0.9899272,0.0002455791,0.00007256204,0.000029103216,0.000060151575,0.0006514946,0.0029153358],"genre_scores_gemma":[0.630657,0.0005337938,0.3615981,0.00028212738,0.00009173816,0.00031143666,0.0002295709,0.00020386142,0.0060923505],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997209,0.000100483056,0.000016075646,0.000064081716,0.00006884366,0.000029566909],"domain_scores_gemma":[0.99931157,0.00036042774,0.0000783715,0.00011040648,0.000074553805,0.0000645862],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000637539,0.0010067784,0.0006190591,0.00022759367,0.00029073047,0.0007924768,0.0014369676,0.00091623294,0.004467667],"category_scores_gemma":[0.002716118,0.0004008099,0.00051409076,0.0002461357,0.0010380697,0.0012861534,0.0015227665,0.0026849485,0.0006475075],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000116809075,0.000102459024,0.00069610565,0.00015829697,0.00007477199,0.00010005359,0.000098574754,0.7695029,0.0035704533,0.1325376,0.003398571,0.08964347],"study_design_scores_gemma":[0.000010666376,0.000025294677,0.000029649651,0.000009672192,0.000005551562,0.0000098657065,0.0000031425159,0.96116215,0.00069226825,0.0364928,0.0015537811,0.0000052352416],"about_ca_topic_score_codex":0.0024028723,"about_ca_topic_score_gemma":0.003583369,"teacher_disagreement_score":0.004467667,"about_ca_system_score_codex":0.0008602045,"about_ca_system_score_gemma":0.0010165448,"threshold_uncertainty_score":0.014945805},"labels":[],"label_agreement":null},{"id":"W2964363500","doi":"10.11159/cist19.118","title":"LongiControl: A New Reinforcement Learning Environment","year":2019,"lang":"en","type":"article","venue":"Proceedings of the World Congress on Electrical Engineering and Computer Systems and Science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Reinforcement learning; Computer science; Human–computer interaction; Artificial intelligence","score_opus":0.005112510692350032,"score_gpt":0.18142821253822908,"score_spread":0.17631570184587905,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964363500","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0035493686,0.00030995632,0.9644152,0.00061414443,0.00025892336,0.00013389833,0.0002993255,0.017250804,0.01316837],"genre_scores_gemma":[0.16806352,0.0005223612,0.79802847,0.0014564068,0.00022124388,0.00093064224,0.0010158243,0.0028509798,0.026910583],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992181,0.00024284204,0.000038170816,0.00015904901,0.0002721466,0.00006970778],"domain_scores_gemma":[0.9990833,0.00042681035,0.00007694677,0.00013159565,0.0000946253,0.00018664385],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012029026,0.000706442,0.00067755644,0.00048050281,0.0005253339,0.0015929003,0.0035521344,0.0012520833,0.019276407],"category_scores_gemma":[0.003154702,0.00043502948,0.00066640595,0.000290139,0.0010124694,0.001999848,0.0038474093,0.0030644713,0.0048640487],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012987655,0.0008816097,0.0016224703,0.00044859725,0.00011326533,0.0007388533,0.0004398656,0.27694505,0.013516574,0.19414335,0.06977466,0.44007698],"study_design_scores_gemma":[0.00023735574,0.00019936528,0.00014025319,0.00005069998,0.000021042222,0.0001945235,0.000024119461,0.80565464,0.0051866425,0.06757115,0.120670594,0.000049620638],"about_ca_topic_score_codex":0.0010689339,"about_ca_topic_score_gemma":0.0018623972,"teacher_disagreement_score":0.019276407,"about_ca_system_score_codex":0.00066289224,"about_ca_system_score_gemma":0.0013014227,"threshold_uncertainty_score":0.06448603},"labels":[],"label_agreement":null},{"id":"W2964469479","doi":"","title":"Understanding the Relation Between Maximum-Entropy Inverse Reinforcement Learning and Behaviour Cloning.","year":2019,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Principle of maximum entropy; Cloning (programming); Artificial intelligence; Computer science; Relation (database); Inverse; Entropy (arrow of time); Mathematics; Machine learning; Data mining; Physics; Thermodynamics","score_opus":0.0945946462711303,"score_gpt":0.3266313768676395,"score_spread":0.2320367305965092,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964469479","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.053229,0.00064983784,0.93418616,0.0013381786,0.00007746276,0.000038412876,0.00006988194,0.00015591958,0.0102550415],"genre_scores_gemma":[0.9333039,0.00033418907,0.062095318,0.00021851185,0.000050846214,0.00007349517,0.00008371508,0.000060232644,0.0037797522],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9992329,0.0003098939,0.000040437615,0.00017573008,0.00015662942,0.00008449442],"domain_scores_gemma":[0.9930681,0.004860662,0.00075286895,0.0006064227,0.00043697905,0.0002749385],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014230154,0.0003764164,0.0005857456,0.00049569755,0.00041306266,0.0014669886,0.0015092813,0.0013724674,0.003745248],"category_scores_gemma":[0.014884394,0.00051129813,0.00055554823,0.00037660825,0.0024310749,0.00373569,0.0015809903,0.0022033656,0.00033482845],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012229919,0.00012413583,0.0028785788,0.000169793,0.000091964066,0.00014628294,0.00032357866,0.27294496,0.0046889535,0.6668865,0.0013042123,0.050318804],"study_design_scores_gemma":[0.00000820924,0.00003634731,0.0006175287,0.00001546159,0.000010511159,0.000044491495,0.000023209159,0.5136353,0.0006543813,0.48449287,0.000450441,0.000011368439],"about_ca_topic_score_codex":0.002052657,"about_ca_topic_score_gemma":0.0018385856,"teacher_disagreement_score":0.003745248,"about_ca_system_score_codex":0.0010883678,"about_ca_system_score_gemma":0.0008043373,"threshold_uncertainty_score":0.012529135},"labels":[],"label_agreement":null},{"id":"W2964599462","doi":"10.1609/aaai.v33i01.33019955","title":"Learning Options with Interest Functions","year":2019,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Task (project management); Function (biology); State space; Plan (archaeology); Space (punctuation); Architecture; State (computer science); Artificial intelligence; Theoretical computer science; Mathematics; Algorithm; Economics","score_opus":0.06783980109864947,"score_gpt":0.2736481056250446,"score_spread":0.20580830452639515,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964599462","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028210055,0.00019966968,0.96832824,0.00039805326,0.000025308475,0.000032271924,0.000045975867,0.00025929796,0.0025012193],"genre_scores_gemma":[0.8470986,0.00024161088,0.14760374,0.00017285587,0.000039592145,0.00017845894,0.00015789458,0.0001075315,0.0043997946],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99911207,0.000429967,0.00004131939,0.00016959586,0.00015816923,0.00008884564],"domain_scores_gemma":[0.99743974,0.001760954,0.00021053661,0.00022274835,0.00019925907,0.00016670772],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021190695,0.00079081644,0.00082563004,0.0004992718,0.00035834796,0.0013730518,0.0013277549,0.0014117003,0.0023057745],"category_scores_gemma":[0.00884419,0.0006226989,0.00075584376,0.0003616416,0.001524165,0.0035774473,0.0020016094,0.0022795994,0.00041066902],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001535919,0.000053095177,0.0013816694,0.00007931623,0.00006262558,0.00014142237,0.00017224293,0.8000362,0.0017087858,0.14787562,0.0012894005,0.047046088],"study_design_scores_gemma":[0.000010715893,0.000019513873,0.00008662363,0.000009543528,0.000005522459,0.000012406008,0.000011235398,0.9308272,0.00041749867,0.06812515,0.00046766616,0.00000699117],"about_ca_topic_score_codex":0.0016375515,"about_ca_topic_score_gemma":0.0018080858,"teacher_disagreement_score":0.0023057745,"about_ca_system_score_codex":0.0010056951,"about_ca_system_score_gemma":0.0008866594,"threshold_uncertainty_score":0.011206806},"labels":[],"label_agreement":null},{"id":"W2964760189","doi":"","title":"Epsilon-BMC: A Bayesian Ensemble Approach to Epsilon-Greedy Exploration in Model-Free Reinforcement Learning","year":2020,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Bayesian probability; Computer science; Perspective (graphical); Mathematical optimization; Monotone polygon; Convergence (economics); Greedy algorithm; Simulated annealing; Bellman equation; Q-learning; Function (biology); Artificial intelligence; Algorithm; Mathematics","score_opus":0.11597923760732981,"score_gpt":0.19311238480803547,"score_spread":0.07713314720070566,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964760189","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003993938,0.00009208753,0.99403775,0.00012447708,0.000022764978,0.00003575724,0.00002606416,0.0004089334,0.0012582054],"genre_scores_gemma":[0.52147126,0.0002294495,0.4726449,0.00047048472,0.00011210633,0.0005238735,0.0002457822,0.00042042983,0.003881712],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983779,0.00069763866,0.00006542515,0.00024299537,0.00044788842,0.0001681346],"domain_scores_gemma":[0.9967998,0.0020247777,0.000279172,0.0003059048,0.00039880807,0.00019154063],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028385713,0.0014270099,0.0020675194,0.0010452076,0.0006846286,0.0012074623,0.0035402675,0.0018936465,0.0036936847],"category_scores_gemma":[0.009108119,0.0010052812,0.0008545765,0.00088086637,0.0012894579,0.0018406687,0.0030785247,0.0030813387,0.00069716584],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009141703,0.00007839937,0.0005061624,0.00006207735,0.00006499034,0.000044836554,0.0000611681,0.912171,0.00082703994,0.022662094,0.0016386976,0.061792143],"study_design_scores_gemma":[0.0000066866014,0.000018720766,0.000024963461,0.000005512342,0.000004570794,0.0000075452563,0.0000024634105,0.9939336,0.00017517894,0.005573579,0.00024316505,0.000003981391],"about_ca_topic_score_codex":0.0053123767,"about_ca_topic_score_gemma":0.008200532,"teacher_disagreement_score":0.0053123767,"about_ca_system_score_codex":0.0014847508,"about_ca_system_score_gemma":0.0022771854,"threshold_uncertainty_score":0.015011966},"labels":[],"label_agreement":null},{"id":"W2964855005","doi":"10.1609/aiide.v15i1.5220","title":"On Hard Exploration for Reinforcement Learning: A Case Study in Pommerman","year":2019,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; SAFER; Benchmark (surveying); Pruning; Computer science; Domain (mathematical analysis); Artificial intelligence; Machine learning; Computer security; Mathematics","score_opus":0.10090642247022633,"score_gpt":0.3290656216999939,"score_spread":0.2281591992297676,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964855005","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.50216156,0.002515132,0.45115903,0.006194767,0.00013488652,0.000348254,0.0004827948,0.0007762083,0.036227383],"genre_scores_gemma":[0.9294494,0.00037564852,0.06640702,0.0002634985,0.00003929119,0.00014033649,0.00012810054,0.00009220842,0.0031045275],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983986,0.00095005496,0.000046324218,0.00016162774,0.00024362784,0.00019981465],"domain_scores_gemma":[0.9862834,0.012001315,0.00036326176,0.0005714763,0.00026124448,0.0005192549],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003639561,0.0007804218,0.0011002064,0.0005978565,0.0013592552,0.0011907388,0.001438384,0.002866841,0.00372086],"category_scores_gemma":[0.016566884,0.00035164558,0.0009028999,0.00071551156,0.0028489665,0.0024553218,0.0021901806,0.0029215447,0.00021508637],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030936496,0.00029807969,0.0022270519,0.00023935095,0.00006950188,0.0007869217,0.00049677433,0.8515396,0.0007924518,0.11785568,0.0024569049,0.022928275],"study_design_scores_gemma":[0.00012838001,0.00015032319,0.00045629294,0.00003494964,0.000013392821,0.00012756852,0.00016420381,0.86229575,0.0008350469,0.13314077,0.0026346564,0.000018587867],"about_ca_topic_score_codex":0.0037044273,"about_ca_topic_score_gemma":0.0052624894,"teacher_disagreement_score":0.00372086,"about_ca_system_score_codex":0.0013019863,"about_ca_system_score_gemma":0.0012850955,"threshold_uncertainty_score":0.019248068},"labels":[],"label_agreement":null},{"id":"W2965050397","doi":"10.24963/ijcai.2019/434","title":"On Principled Entropy Exploration in Policy Optimization","year":2019,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Benchmark (surveying); Entropy (arrow of time); Mathematical optimization; Optimization problem; Convergence (economics); Set (abstract data type); Monotonic function; Constrained optimization problem; Multi-objective optimization; Descent (aeronautics); Machine learning; Algorithm; Mathematics; Economics; Engineering","score_opus":0.013976204612210875,"score_gpt":0.254103345513721,"score_spread":0.2401271409015101,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2965050397","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011399888,0.00049316674,0.9845659,0.0003375122,0.000030469953,0.000041191026,0.000016102049,0.00013405693,0.0029816595],"genre_scores_gemma":[0.7620971,0.0010021225,0.23187847,0.00043411175,0.00015773797,0.0004049047,0.00006999146,0.00020585391,0.003749685],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.998615,0.00078366516,0.000048679878,0.00014657913,0.00031942147,0.00008667416],"domain_scores_gemma":[0.99541557,0.0037449882,0.00024378087,0.00025845863,0.00021899816,0.00011818246],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030820714,0.0011986039,0.0010249037,0.00066751643,0.00048099682,0.0008929384,0.0008846472,0.0010763839,0.0015302933],"category_scores_gemma":[0.0105991475,0.0005917735,0.00056612573,0.00048989773,0.0027711368,0.0018458308,0.002573563,0.0017182637,0.00032294812],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010987737,0.000053465454,0.00055999827,0.00015952128,0.000051323506,0.00006996744,0.0001362428,0.87641764,0.0023957773,0.0840744,0.00071641535,0.035255376],"study_design_scores_gemma":[0.000016997772,0.00007423154,0.00007068957,0.000019889454,0.000006218312,0.000022653183,0.000009112258,0.9612362,0.00058700086,0.037354622,0.0005944028,0.000008117466],"about_ca_topic_score_codex":0.0009792205,"about_ca_topic_score_gemma":0.0008775272,"teacher_disagreement_score":0.0030820714,"about_ca_system_score_codex":0.00079309894,"about_ca_system_score_gemma":0.0013392633,"threshold_uncertainty_score":0.016299725},"labels":[],"label_agreement":null},{"id":"W2965212561","doi":"","title":"Neural Graph Evolution: Towards Efficient Automatic Robot Design","year":2019,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Robot; Graph; Artificial intelligence; Artificial neural network; Theoretical computer science; Machine learning","score_opus":0.05089747332137666,"score_gpt":0.3173870093774977,"score_spread":0.266489536056121,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2965212561","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0123368185,0.00028451515,0.9837779,0.0001877398,0.000029422417,0.000051790394,0.000041978834,0.0006454199,0.0026444467],"genre_scores_gemma":[0.38192153,0.00043888838,0.61238277,0.0003225056,0.000038784157,0.00031326612,0.00025037196,0.00032890818,0.0040028812],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995654,0.0001064259,0.000018349076,0.0000946976,0.00016965743,0.00004544776],"domain_scores_gemma":[0.99924785,0.00040051944,0.00009405468,0.000087994864,0.00013440911,0.000035154382],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006796625,0.0009866758,0.0007902243,0.00086592487,0.00037116086,0.00056045456,0.0012878776,0.0010873667,0.0024254909],"category_scores_gemma":[0.002234018,0.00060017727,0.00083968876,0.0006374503,0.0009861783,0.0008744169,0.0009811224,0.0011761497,0.00049158663],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000026864984,0.00003387823,0.00043282783,0.000085187654,0.000026829652,0.00005514682,0.000043956483,0.8910605,0.003788872,0.015021123,0.0014992934,0.087925546],"study_design_scores_gemma":[0.0000060337443,0.000014153771,0.00003803579,0.0000048362426,0.0000035734229,0.000010849114,0.0000041277117,0.99285203,0.00040736163,0.005891312,0.0007652099,0.0000025151426],"about_ca_topic_score_codex":0.003939097,"about_ca_topic_score_gemma":0.005183486,"teacher_disagreement_score":0.003939097,"about_ca_system_score_codex":0.0009223709,"about_ca_system_score_gemma":0.0013464687,"threshold_uncertainty_score":0.0081140995},"labels":[],"label_agreement":null},{"id":"W2966234803","doi":"10.24963/ijcai.2019/445","title":"Hill Climbing on Value Estimates for Search-control in Dyna","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Toronto; Huawei Technologies (Canada); University of Alberta","funders":"","keywords":"Hill climbing; Climbing; Reinforcement learning; Trajectory; Computer science; Bellman equation; Langevin dynamics; Value (mathematics); Sampling (signal processing); Artificial intelligence; Mathematical optimization; Mathematics; Machine learning; Statistics; Engineering","score_opus":0.03048606102777367,"score_gpt":0.29886017952237093,"score_spread":0.26837411849459725,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2966234803","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033080623,0.00015646548,0.9639745,0.00022891015,0.000028113242,0.00005383474,0.000041302395,0.00036928357,0.0020670332],"genre_scores_gemma":[0.7687293,0.00015670292,0.22769886,0.0001540691,0.000026320038,0.0002756561,0.00013162429,0.0001990715,0.0026282766],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992698,0.00034262653,0.000038393675,0.00012082794,0.00016571765,0.000062701394],"domain_scores_gemma":[0.99624324,0.0026028487,0.00033603553,0.00031847536,0.00033810956,0.00016133128],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018052489,0.0007023618,0.0010722685,0.0006219122,0.00067012315,0.0010551985,0.0011131178,0.00088990905,0.002251586],"category_scores_gemma":[0.011204827,0.0006455444,0.00053027394,0.00042101127,0.0017589696,0.0013632177,0.001263374,0.0016593672,0.00032283485],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006137623,0.000030955023,0.0009306394,0.00005052539,0.000025457932,0.000049492977,0.000091934766,0.9411657,0.00083263166,0.041540667,0.00058625767,0.014634299],"study_design_scores_gemma":[0.000007347784,0.000016037668,0.00006941099,0.0000068416875,0.0000024445699,0.000007334685,0.0000053986164,0.9892898,0.00024873158,0.010111808,0.00023062098,0.0000042796305],"about_ca_topic_score_codex":0.0050603775,"about_ca_topic_score_gemma":0.005857378,"teacher_disagreement_score":0.0050603775,"about_ca_system_score_codex":0.0014605236,"about_ca_system_score_gemma":0.0015462472,"threshold_uncertainty_score":0.010596871},"labels":[],"label_agreement":null},{"id":"W2966675798","doi":"10.24963/ijcai.2019/506","title":"Planning with Expectation Models","year":2019,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Convergence (economics); Parametrization (atmospheric modeling); Mathematical optimization; Bellman equation; Function (biology); State (computer science); Artificial intelligence; Mathematics; Algorithm","score_opus":0.016672741621003975,"score_gpt":0.2247960668066918,"score_spread":0.20812332518568782,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2966675798","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032071632,0.00013113405,0.99436706,0.00020866918,0.000016857733,0.000025008001,0.000058678335,0.00022622675,0.00175914],"genre_scores_gemma":[0.6222598,0.00069379364,0.36887038,0.00041130127,0.000089927446,0.00046438855,0.0004828216,0.00021393738,0.006513711],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998125,0.0008385746,0.00010561379,0.0003750243,0.00039320355,0.00016243849],"domain_scores_gemma":[0.99573624,0.0032621305,0.00029340756,0.0003185079,0.00026709918,0.00012253907],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019513866,0.0012026901,0.0012458201,0.0005430862,0.0004415317,0.0016920777,0.0019138269,0.0014640132,0.0048889536],"category_scores_gemma":[0.010195624,0.00067396887,0.0011816708,0.0007982532,0.0015063579,0.003436753,0.0020088311,0.0027246082,0.00075714145],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010422318,0.000049654744,0.00047589582,0.00011369844,0.000039118524,0.000099304416,0.00011486416,0.7760279,0.0005929056,0.18755639,0.001426262,0.03339976],"study_design_scores_gemma":[0.000016312488,0.000030761228,0.00004785111,0.000013428029,0.000008561177,0.000025323896,0.000009472781,0.916645,0.0003132658,0.081954725,0.00092569814,0.000009521326],"about_ca_topic_score_codex":0.0046702954,"about_ca_topic_score_gemma":0.004702372,"teacher_disagreement_score":0.0048889536,"about_ca_system_score_codex":0.0013688086,"about_ca_system_score_gemma":0.0016363178,"threshold_uncertainty_score":0.016355157},"labels":[],"label_agreement":null},{"id":"W2968035913","doi":"10.1109/med.2019.8798515","title":"Lamarckian Inheritance in Neuromodulated Multiobjective Evolutionary Neurocontrollers","year":2019,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Neuroevolution; Artificial intelligence; Inheritance (genetic algorithm); Evolutionary algorithm; Computer science; Traverse; Artificial neural network; Biology","score_opus":0.006355943929348623,"score_gpt":0.2070230902286131,"score_spread":0.20066714629926447,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2968035913","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12651083,0.00039687322,0.8661903,0.00013230805,0.00004717422,0.000066396184,0.000024110004,0.0004508448,0.0061812024],"genre_scores_gemma":[0.8242979,0.00012609031,0.1715278,0.00011045375,0.000011656526,0.00014222255,0.000030395677,0.000036408503,0.0037169978],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997764,0.000056150166,0.000013287601,0.00004182687,0.000092250244,0.000020080464],"domain_scores_gemma":[0.99967027,0.00010578012,0.00007873876,0.000039725048,0.000083737796,0.000021730899],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00045931424,0.00037355942,0.00033855587,0.00032239594,0.00026794124,0.00041090042,0.0011293353,0.00052831974,0.0007227207],"category_scores_gemma":[0.001027039,0.00020635064,0.0003179656,0.00018762007,0.00043858917,0.00038467234,0.00053036446,0.0005562807,0.00013239738],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003644614,0.000073095725,0.00087179185,0.00006334734,0.00006385283,0.00012598051,0.000070610586,0.87727106,0.029984321,0.008610093,0.00031122053,0.08251823],"study_design_scores_gemma":[0.000008415612,0.00006355767,0.00018652975,0.0000062756217,0.000007425229,0.00005410494,0.0000048283077,0.9942204,0.0034001258,0.001186283,0.0008551992,0.000006850463],"about_ca_topic_score_codex":0.0012528987,"about_ca_topic_score_gemma":0.0016354526,"teacher_disagreement_score":0.0012528987,"about_ca_system_score_codex":0.0006208893,"about_ca_system_score_gemma":0.00047496613,"threshold_uncertainty_score":0.0045048594},"labels":[],"label_agreement":null},{"id":"W2970384648","doi":"10.48550/arxiv.1912.04226","title":"Unsupervised Curricula for Visual Meta-Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Reinforcement learning; Unsupervised learning; Artificial intelligence; Machine learning; Meta learning (computer science); Cluster analysis; Task (project management); Discriminative model; Trajectory","score_opus":0.06914127001221135,"score_gpt":0.2036564058526367,"score_spread":0.13451513584042535,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2970384648","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017657897,0.0001272085,0.9791834,0.00022993925,0.000020134874,0.00007646219,0.0000558215,0.0005882589,0.0020609922],"genre_scores_gemma":[0.6907538,0.00017783446,0.30362168,0.00020947806,0.000041546788,0.00056684134,0.00022218667,0.00020962833,0.004196979],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994137,0.00022275072,0.000030206296,0.00017056233,0.00009849423,0.000064224136],"domain_scores_gemma":[0.99790835,0.0010978152,0.00019623956,0.0004010027,0.0002470426,0.00014946441],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013540211,0.00090933713,0.0007218548,0.0004626646,0.0004831971,0.0008855378,0.0019644222,0.0010309775,0.0032171125],"category_scores_gemma":[0.007063257,0.00056425395,0.0006060272,0.00038041745,0.0013806636,0.0018042481,0.0019672315,0.0020115967,0.00067412364],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012951705,0.00019699387,0.0018222032,0.00015729465,0.00006805388,0.00006256158,0.0002683374,0.7790239,0.0046478547,0.070870005,0.0019964462,0.14075685],"study_design_scores_gemma":[0.000019014646,0.000039200433,0.00012567716,0.000013340227,0.000005690446,0.0000107602045,0.000011259533,0.96722335,0.001053072,0.030658958,0.000833283,0.0000063988555],"about_ca_topic_score_codex":0.0016364744,"about_ca_topic_score_gemma":0.0032129027,"teacher_disagreement_score":0.0032171125,"about_ca_system_score_codex":0.001360805,"about_ca_system_score_gemma":0.001444407,"threshold_uncertainty_score":0.010762274},"labels":[],"label_agreement":null},{"id":"W2970483889","doi":"10.48550/arxiv.1911.04448","title":"Real-Time Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Open Philanthropy Project","keywords":"Reinforcement learning; Markov decision process; Computer science; Action selection; Computation; Artificial intelligence; State (computer science); Action (physics); Markov process; Mathematical optimization; Machine learning; Algorithm; Mathematics","score_opus":0.008090322237890085,"score_gpt":0.22149011163154358,"score_spread":0.2133997893936535,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2970483889","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016185815,0.0012461684,0.96664166,0.00058518053,0.00025301398,0.00008170822,0.0000778394,0.0012587109,0.013669981],"genre_scores_gemma":[0.7911433,0.0011458191,0.18478027,0.00035275894,0.000136248,0.00021401586,0.00023177848,0.0001735156,0.021822244],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999124,0.00028213166,0.000046330562,0.00021891991,0.00024603147,0.00008250719],"domain_scores_gemma":[0.9986499,0.0006982684,0.00012950691,0.0001530338,0.00028510342,0.00008422973],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011696556,0.0007917386,0.00071302097,0.00029192172,0.00031516128,0.001246099,0.0012286964,0.0009188627,0.0062331012],"category_scores_gemma":[0.004235407,0.00026351758,0.0004496548,0.00028561568,0.000775684,0.0010812578,0.00091534643,0.0014057297,0.0011857491],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002625767,0.00015008058,0.0011391704,0.00029476607,0.000091599475,0.00019954686,0.00013855955,0.7322657,0.005956901,0.069099195,0.0055527305,0.18484925],"study_design_scores_gemma":[0.00002697593,0.000056546538,0.0001461117,0.000014601565,0.000011435178,0.000042362673,0.00001376678,0.97579646,0.001369457,0.017043723,0.0054680393,0.000010467642],"about_ca_topic_score_codex":0.003052456,"about_ca_topic_score_gemma":0.0032278341,"teacher_disagreement_score":0.0062331012,"about_ca_system_score_codex":0.00095312204,"about_ca_system_score_gemma":0.0012603578,"threshold_uncertainty_score":0.02085179},"labels":[],"label_agreement":null},{"id":"W2970634945","doi":"","title":"Maximum Entropy Monte-Carlo Planning","year":2019,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Softmax function; Computer science; Mathematical optimization; Monte Carlo method; Principle of maximum entropy; Entropy (arrow of time); Convergence (economics); Monte Carlo tree search; Rate of convergence; Mathematics; Algorithm; Artificial intelligence; Artificial neural network; Statistics","score_opus":0.015684304278766808,"score_gpt":0.24203487129328252,"score_spread":0.22635056701451572,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2970634945","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006118774,0.000102566235,0.9901484,0.0001781429,0.000024357829,0.000048971222,0.000040570703,0.00035959747,0.0029785053],"genre_scores_gemma":[0.4669686,0.00017406298,0.528085,0.000208067,0.00006347482,0.00036655867,0.0002068084,0.0001639627,0.0037635593],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987871,0.0004479472,0.00006146169,0.00018478493,0.0004068967,0.00011194858],"domain_scores_gemma":[0.99520195,0.0035836473,0.0002913493,0.00044231335,0.00031499335,0.00016569471],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022470239,0.00076479436,0.0012141377,0.00083794544,0.00057489134,0.0009870332,0.0017512084,0.0011314217,0.004056892],"category_scores_gemma":[0.009607419,0.0006551463,0.0007917506,0.00097308436,0.0015738831,0.0014161997,0.0017334644,0.0016972545,0.00063900324],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000053244854,0.000025949073,0.00034421406,0.000034433142,0.000022879873,0.000036742225,0.000033762088,0.93034077,0.00036330288,0.041556265,0.00075312186,0.026435293],"study_design_scores_gemma":[0.000004480021,0.000005724883,0.00001692791,0.0000026960365,0.0000016362444,0.0000052257355,0.0000011706799,0.9912503,0.00012143717,0.008362213,0.00022668485,0.0000015285009],"about_ca_topic_score_codex":0.004654962,"about_ca_topic_score_gemma":0.005778416,"teacher_disagreement_score":0.004654962,"about_ca_system_score_codex":0.001694972,"about_ca_system_score_gemma":0.0021728117,"threshold_uncertainty_score":0.01357168},"labels":[],"label_agreement":null},{"id":"W2970673985","doi":"","title":"Learning Reward Machines for Partially Observable Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":67,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Observable; Computer science; Set (abstract data type); Artificial intelligence; Task (project management); Function (biology); Optimization problem; Learning automata; Representation (politics); Decomposition; Automaton; Mathematical optimization; Mathematics; Algorithm","score_opus":0.021000865689137876,"score_gpt":0.2534669585508525,"score_spread":0.23246609286171463,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2970673985","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003930284,0.00024043898,0.99349046,0.00024041148,0.000036019188,0.000044752407,0.0000463026,0.00052244007,0.0014488812],"genre_scores_gemma":[0.56991696,0.00059143413,0.4243037,0.00036589804,0.00014576144,0.0008673118,0.0002886441,0.00022069496,0.0032996999],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988599,0.00057880866,0.00006771642,0.0002061294,0.00020505769,0.00008234437],"domain_scores_gemma":[0.99555296,0.0033537194,0.00028026907,0.00036249176,0.0003376862,0.0001128223],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019453714,0.0010812506,0.0012639469,0.00045666922,0.00039145115,0.0011933055,0.0013086742,0.0013708677,0.0035795302],"category_scores_gemma":[0.009478923,0.00048667623,0.00064921513,0.0004970412,0.0014733108,0.0016516156,0.0011680619,0.003176733,0.0006099307],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000061439896,0.00005809307,0.000379242,0.00010719877,0.000036177396,0.00004471012,0.000053706248,0.84229773,0.00071420986,0.11113393,0.001594227,0.043519422],"study_design_scores_gemma":[0.000010182393,0.000012301267,0.000024385701,0.000006967145,0.0000027943481,0.0000049160344,0.0000026492564,0.9647503,0.00018586595,0.03450331,0.00049279263,0.0000035004618],"about_ca_topic_score_codex":0.0014463402,"about_ca_topic_score_gemma":0.0018338809,"teacher_disagreement_score":0.0035795302,"about_ca_system_score_codex":0.001200085,"about_ca_system_score_gemma":0.0012643383,"threshold_uncertainty_score":0.011974692},"labels":[],"label_agreement":null},{"id":"W2970912723","doi":"","title":"Gossip-based Actor-Learner Architectures for Deep Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Gossip; Asynchronous communication; Computer science; Reinforcement learning; Scalability; Distributed computing; Synchronization (alternating current); Gossip protocol; Artificial intelligence; Computer network","score_opus":0.03552543995273192,"score_gpt":0.18424288550916965,"score_spread":0.14871744555643773,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2970912723","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03846586,0.00022051502,0.95723957,0.00025833363,0.00006331276,0.00004197756,0.000039730596,0.00070409285,0.0029666352],"genre_scores_gemma":[0.9262322,0.00013279382,0.0703412,0.00010124036,0.000032991444,0.00011724784,0.0000619987,0.00006738565,0.0029128485],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996332,0.0001401637,0.000018710356,0.00007715951,0.00008029474,0.000050403814],"domain_scores_gemma":[0.99899966,0.0005684245,0.00010503106,0.00011619272,0.0001260476,0.00008471731],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009806219,0.00064359145,0.0006437381,0.00021545768,0.00044012794,0.00057123427,0.0012657546,0.0007707194,0.002002971],"category_scores_gemma":[0.0025339446,0.00029280587,0.0003161619,0.00021028974,0.0010031612,0.0007755255,0.0011410041,0.0014213598,0.00039062134],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010977287,0.000036735306,0.00039070926,0.000038832975,0.00003269606,0.000046974958,0.000054033255,0.9618142,0.0025555778,0.015726052,0.0007166102,0.018477788],"study_design_scores_gemma":[0.0000063865823,0.000016597905,0.000021093449,0.0000013933869,0.0000022237696,0.0000034793388,0.000002783949,0.9955213,0.00035167,0.0038516428,0.0002193494,0.0000021219337],"about_ca_topic_score_codex":0.0021904774,"about_ca_topic_score_gemma":0.0031983624,"teacher_disagreement_score":0.0021904774,"about_ca_system_score_codex":0.0007449204,"about_ca_system_score_gemma":0.00087246235,"threshold_uncertainty_score":0.006700635},"labels":[],"label_agreement":null},{"id":"W2971218263","doi":"","title":"SMILe: Scalable Meta Inverse Reinforcement Learning through Context-Conditional Policies","year":2019,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Meta learning (computer science); Scalability; Machine learning; Principle of maximum entropy; Context (archaeology); Task (project management)","score_opus":0.03388311419550946,"score_gpt":0.2615376253999302,"score_spread":0.22765451120442073,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2971218263","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024670122,0.0005273281,0.96716124,0.00035148772,0.00008818153,0.000107339685,0.00012326911,0.004477574,0.002493581],"genre_scores_gemma":[0.84065586,0.00017944098,0.1552905,0.00040475768,0.000055049954,0.00033915925,0.00029469462,0.0003657138,0.0024148175],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99940526,0.00020235087,0.000028720759,0.00014593048,0.00013241936,0.00008534191],"domain_scores_gemma":[0.9982615,0.0010723267,0.00014891606,0.00023611938,0.00015596063,0.0001251785],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015669023,0.00122024,0.0014132385,0.0004488206,0.00040599663,0.00088002667,0.002225241,0.0013979359,0.003678383],"category_scores_gemma":[0.00562928,0.0006953441,0.00071759504,0.00033361415,0.0012060595,0.001376939,0.002256623,0.0024510755,0.000804481],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012394908,0.00010215726,0.0007774272,0.00009720742,0.000061318126,0.00008341508,0.00006066423,0.92714065,0.0019096585,0.0056048213,0.0018332679,0.06220553],"study_design_scores_gemma":[0.000013529026,0.000023391112,0.00003277231,0.000004509436,0.0000035050834,0.0000071131253,0.0000027684132,0.99717665,0.00028992145,0.00223144,0.00021121583,0.0000031706616],"about_ca_topic_score_codex":0.003964559,"about_ca_topic_score_gemma":0.005016447,"teacher_disagreement_score":0.003964559,"about_ca_system_score_codex":0.0008662162,"about_ca_system_score_gemma":0.0017689058,"threshold_uncertainty_score":0.0123054385},"labels":[],"label_agreement":null},{"id":"W2971273232","doi":"10.13140/rg.2.2.12370.50882","title":"The Option Keyboard: Combining Skills in Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Cumulant; Reinforcement learning; Computer science; Premise; Artificial intelligence; Machine learning; Formalism (music); Transfer of learning; Human–computer interaction; Mathematics","score_opus":0.022666124344261605,"score_gpt":0.17229140948524505,"score_spread":0.14962528514098344,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2971273232","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017606925,0.000080778365,0.97933596,0.00017363322,0.000022063643,0.00004051852,0.000021312131,0.00023448456,0.0024842522],"genre_scores_gemma":[0.5806797,0.00015339455,0.41572526,0.0001410019,0.00003796086,0.00024281735,0.000047698577,0.000089631394,0.0028825337],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99906176,0.00048135716,0.00004097013,0.00016227539,0.00017606848,0.000077669916],"domain_scores_gemma":[0.9984162,0.000980119,0.0001264844,0.00021557437,0.00010195479,0.00015959435],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018712984,0.00066008087,0.0005961992,0.00048211758,0.0004845042,0.0011099994,0.0013063131,0.001024938,0.004216657],"category_scores_gemma":[0.0057374476,0.0003858618,0.000751576,0.00047352535,0.0029280402,0.0030344576,0.0021468,0.0017024641,0.00043288086],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034051444,0.0001870814,0.0016335646,0.00017324582,0.00007552438,0.00032637006,0.000556198,0.3645539,0.0125185205,0.4674414,0.0012135443,0.15098022],"study_design_scores_gemma":[0.00004926792,0.00016815314,0.00023406031,0.000032920918,0.000015741416,0.00006543199,0.000035147186,0.7025837,0.0024718728,0.2917365,0.0025804304,0.000026836005],"about_ca_topic_score_codex":0.0010421323,"about_ca_topic_score_gemma":0.00093942054,"teacher_disagreement_score":0.004216657,"about_ca_system_score_codex":0.0006467643,"about_ca_system_score_gemma":0.00064431137,"threshold_uncertainty_score":0.014106154},"labels":[],"label_agreement":null},{"id":"W2971484784","doi":"10.1609/aaai.v34i04.5784","title":"Fixed-Horizon Temporal Difference Methods for Stable Reinforcement Learning","year":2020,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Huawei Technologies (Canada); University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Vector Institute; Alberta Innovates; DeepMind","keywords":"Reinforcement learning; Bellman equation; Horizon; Temporal difference learning; Function (biology); Time horizon; Value (mathematics); Stability (learning theory); Computer science; Mathematics; Mathematical optimization; Artificial intelligence; Machine learning","score_opus":0.137803606700458,"score_gpt":0.36583601233052376,"score_spread":0.22803240563006577,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2971484784","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002313752,0.00020904683,0.9959947,0.000101753016,0.00004352049,0.000018102808,0.000014014725,0.0000832907,0.001221762],"genre_scores_gemma":[0.5148572,0.00062327686,0.47777,0.00030451413,0.000103872146,0.0003541287,0.00010613056,0.00020774835,0.005673142],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994004,0.00024445605,0.0000293931,0.00010512631,0.00017122581,0.000049409344],"domain_scores_gemma":[0.99714124,0.002103798,0.0002088576,0.00015068245,0.00029490096,0.00010049272],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001982418,0.0008347128,0.0008894865,0.0003808095,0.0003855237,0.00094686635,0.0017546888,0.0011197525,0.003985907],"category_scores_gemma":[0.0069301175,0.00041784663,0.00065450167,0.00043591243,0.0012793556,0.0013667464,0.0013115535,0.0023018967,0.0005145989],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000086453685,0.000065863074,0.00047272316,0.00015072958,0.000048775622,0.000058640147,0.00008713765,0.8343075,0.0017040113,0.115035795,0.0011344836,0.046847884],"study_design_scores_gemma":[0.000008464381,0.00001550128,0.000018433153,0.000008116852,0.000003128749,0.0000060251364,0.0000026726004,0.9816613,0.00025996365,0.017441828,0.00057096774,0.0000035829096],"about_ca_topic_score_codex":0.0035379112,"about_ca_topic_score_gemma":0.0023885893,"teacher_disagreement_score":0.003985907,"about_ca_system_score_codex":0.00130499,"about_ca_system_score_gemma":0.0011450376,"threshold_uncertainty_score":0.013334215},"labels":[],"label_agreement":null},{"id":"W2972435663","doi":"10.65109/hlnw2204","title":"Safe Policy Improvement with an Estimated Baseline Policy","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Microsoft (Canada)","funders":"","keywords":"Baseline (sea); Reinforcement learning; Bootstrapping (finance); Computer science; Variance (accounting); Control (management); Machine learning; Artificial intelligence; Econometrics; Mathematics; Economics; Political science; Accounting","score_opus":0.03920345239348902,"score_gpt":0.31655273234008885,"score_spread":0.27734927994659986,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2972435663","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029711846,0.00047721557,0.96401215,0.0003728943,0.00009804301,0.00009711578,0.00011436663,0.0028209095,0.00229558],"genre_scores_gemma":[0.7126206,0.00020159675,0.28375,0.0004276767,0.00006830563,0.00021266963,0.00041963212,0.00038817106,0.0019114057],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99727553,0.000763748,0.00015616199,0.00085187383,0.0006183012,0.00033436556],"domain_scores_gemma":[0.9913292,0.005190293,0.0006094324,0.0015367563,0.0009998465,0.00033436844],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004450984,0.0017486587,0.0016399589,0.0007854746,0.00066186645,0.0012874955,0.0018683448,0.0018598569,0.0026690022],"category_scores_gemma":[0.02220611,0.00065659214,0.0007797445,0.0005603595,0.0017706066,0.0022128103,0.0022404918,0.004248686,0.0011790937],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004584149,0.00020148483,0.0019327773,0.00017444145,0.00008391867,0.00011890406,0.00016843744,0.8362374,0.0047585424,0.014218995,0.0027965785,0.13885011],"study_design_scores_gemma":[0.000029974603,0.00011590117,0.00021066701,0.000024993802,0.00001195639,0.000034337714,0.000017695036,0.9866998,0.0024424146,0.009709762,0.00068962824,0.000012811624],"about_ca_topic_score_codex":0.004782636,"about_ca_topic_score_gemma":0.004019497,"teacher_disagreement_score":0.004782636,"about_ca_system_score_codex":0.0014584041,"about_ca_system_score_gemma":0.0045161615,"threshold_uncertainty_score":0.023539305},"labels":[],"label_agreement":null},{"id":"W2975017304","doi":"10.1109/cig.2019.8848040","title":"Using Simple Games to Evaluate Self-Organization Concepts: a Whack-a-mole Case Study","year":2019,"lang":"en","type":"article","venue":"2019 IEEE Conference on Games (CoG)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Generality; Simple (philosophy); Hammer; Computer science; Task (project management); Industrial engineering; Engineering; Psychology; Epistemology; Mechanical engineering; Systems engineering","score_opus":0.05754343499664861,"score_gpt":0.34501022742724996,"score_spread":0.28746679243060136,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2975017304","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92316395,0.00030685763,0.061791684,0.00044052667,0.00007447529,0.0008325138,0.00029720055,0.0002299871,0.0128628835],"genre_scores_gemma":[0.9365687,0.000117625365,0.05960972,0.00009717349,0.000012606651,0.00027434085,0.00019216514,0.00004352973,0.0030841455],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980563,0.0011462457,0.0000995256,0.00017391871,0.0003730948,0.0001509082],"domain_scores_gemma":[0.992718,0.005434216,0.00037916776,0.0005923853,0.0004496573,0.0004264946],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026284682,0.0010917068,0.00074182963,0.0008640054,0.0008082973,0.0013620245,0.0016178726,0.0016310418,0.0032120522],"category_scores_gemma":[0.011206644,0.00027040156,0.00063085527,0.0006840709,0.0015026662,0.002203737,0.0014395912,0.0014612491,0.0003136963],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0034527022,0.009471444,0.025846886,0.0015165334,0.0005716437,0.002382403,0.0028609456,0.65053195,0.022985835,0.12150498,0.006923802,0.1519509],"study_design_scores_gemma":[0.00054500054,0.0023010967,0.0055027907,0.00006432797,0.00008100818,0.00027452136,0.0010587856,0.9268576,0.01570014,0.03962651,0.007914929,0.00007331052],"about_ca_topic_score_codex":0.0067008757,"about_ca_topic_score_gemma":0.008799664,"teacher_disagreement_score":0.0067008757,"about_ca_system_score_codex":0.001457179,"about_ca_system_score_gemma":0.0006049734,"threshold_uncertainty_score":0.013900876},"labels":[],"label_agreement":null},{"id":"W2977481643","doi":"10.1609/aaai.v35i12.17276","title":"Improving Sample Efficiency in Model-Free Reinforcement Learning from Images","year":2021,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":121,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Robustness (evolution); Computer science; Artificial intelligence; Stability (learning theory); Machine learning; Encoder; Representation (politics); Noise (video); Code (set theory); Image (mathematics)","score_opus":0.07127289501225326,"score_gpt":0.2836056152858242,"score_spread":0.21233272027357092,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2977481643","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04095241,0.00017984388,0.9556742,0.00031453578,0.00002435989,0.00005164371,0.00003249415,0.0010664491,0.0017041086],"genre_scores_gemma":[0.8612784,0.00009830592,0.1359755,0.00021622963,0.000028243012,0.00014765744,0.00010819449,0.00024429904,0.0019031663],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991339,0.00036306467,0.00004180688,0.00018450107,0.0001717569,0.000105062405],"domain_scores_gemma":[0.9942808,0.004170848,0.0003821589,0.00065927004,0.00032221392,0.00018482201],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030991547,0.0010846577,0.0015781464,0.00048731704,0.0005415608,0.0010274079,0.0019593542,0.001416929,0.0024105704],"category_scores_gemma":[0.014849336,0.0007995028,0.0005556846,0.00033335536,0.0017855836,0.00218686,0.0019454927,0.0022878,0.0004862301],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022755864,0.00012965086,0.001024837,0.000085478576,0.000039371873,0.000073304655,0.00010217702,0.9321941,0.0023441722,0.016762858,0.0008702454,0.04614608],"study_design_scores_gemma":[0.000013135978,0.000022186074,0.00004434441,0.0000047138637,0.0000027149563,0.0000067879178,0.000003214012,0.99473214,0.0004665226,0.0046178848,0.000083833715,0.0000025481556],"about_ca_topic_score_codex":0.0048994436,"about_ca_topic_score_gemma":0.0049686274,"teacher_disagreement_score":0.0048994436,"about_ca_system_score_codex":0.0012486505,"about_ca_system_score_gemma":0.0016416587,"threshold_uncertainty_score":0.016390085},"labels":[],"label_agreement":null},{"id":"W2977878187","doi":"10.1007/978-3-030-23807-0_27","title":"Intrinsically Motivated Autonomy in Human-Robot Interaction: Human Perception of Predictive Information in Robots","year":2019,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Robot; Perception; Human–computer interaction; Human–robot interaction; Autonomy; Baseline (sea); USable; Artificial intelligence; Competence (human resources); Psychology; Social psychology","score_opus":0.015830707686866598,"score_gpt":0.26033507202411815,"score_spread":0.24450436433725156,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2977878187","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10091023,0.03523254,0.78697264,0.0031639333,0.00064006436,0.000028733031,0.000113209324,0.0005490866,0.072389595],"genre_scores_gemma":[0.9331014,0.006966018,0.046249717,0.00017544303,0.00022430305,0.000040235424,0.00008657754,0.00008668908,0.01306962],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9998859,0.000039029514,0.0000031777179,0.000028461161,0.000032738044,0.00001055008],"domain_scores_gemma":[0.9996933,0.0002165883,0.000022645925,0.000028258208,0.0000218842,0.00001733185],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00023071209,0.00025457086,0.00027988444,0.00012095755,0.0001642506,0.0011111705,0.00050600397,0.0007006043,0.0021582947],"category_scores_gemma":[0.0009275422,0.00019544069,0.00025412708,0.0002809327,0.0011167763,0.0014005863,0.00057588046,0.0009642719,0.00031000166],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020649559,0.00013538773,0.0009502892,0.00056906074,0.00008419301,0.00031695556,0.001326703,0.11809157,0.036201168,0.56052774,0.014363627,0.26722687],"study_design_scores_gemma":[0.000016940354,0.00011459887,0.0024068423,0.00008067945,0.000022372402,0.00025520584,0.0002548579,0.3039293,0.008211258,0.6632134,0.021440072,0.000054525255],"about_ca_topic_score_codex":0.00047673125,"about_ca_topic_score_gemma":0.0004234322,"teacher_disagreement_score":0.0021582947,"about_ca_system_score_codex":0.0003059282,"about_ca_system_score_gemma":0.0002354324,"threshold_uncertainty_score":0.0072202086},"labels":[],"label_agreement":null},{"id":"W2987502141","doi":"10.1609/aaai.v34i04.6027","title":"Gamma-Nets: Generalizing Value Estimation over Timescale","year":2020,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates; Alberta Machine Intelligence Institute; Compute Canada","keywords":"Estimator; Reinforcement learning; Computer science; A priori and a posteriori; Bellman equation; Representation (politics); Function (biology); Value (mathematics); Artificial intelligence; Algorithm; Machine learning; Mathematical optimization; Mathematics; Statistics","score_opus":0.09424233049296664,"score_gpt":0.31212231732304635,"score_spread":0.2178799868300797,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2987502141","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022689054,0.0003960851,0.9733612,0.00040187596,0.000072147486,0.00004923847,0.00017566793,0.0012044,0.0016503222],"genre_scores_gemma":[0.74909496,0.0006920793,0.2440497,0.00054960424,0.00012787744,0.000249509,0.00062069105,0.0003590758,0.0042565414],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99939966,0.00015749187,0.000039053455,0.00018656183,0.00013508766,0.00008211402],"domain_scores_gemma":[0.99775106,0.0013725082,0.00026685058,0.0002253468,0.00025335455,0.0001308166],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021491363,0.0015829562,0.0011112954,0.00079645164,0.00046156315,0.0012660252,0.0024697199,0.0015935161,0.00231283],"category_scores_gemma":[0.007778883,0.0008605263,0.0010128287,0.00060570205,0.0014525979,0.0034730448,0.0020500908,0.0033892775,0.00048357103],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008299572,0.00003008702,0.0012644123,0.00004460598,0.000035885758,0.00005779808,0.00006400648,0.9384231,0.0009370685,0.012917353,0.0011489829,0.044993788],"study_design_scores_gemma":[0.0000050533995,0.000012197263,0.000059376413,0.0000069420203,0.000004220977,0.000006215806,0.0000044295325,0.9863127,0.00029453577,0.0130227525,0.00026770207,0.000004013421],"about_ca_topic_score_codex":0.013551806,"about_ca_topic_score_gemma":0.012117948,"teacher_disagreement_score":0.013551806,"about_ca_system_score_codex":0.0021609932,"about_ca_system_score_gemma":0.0016301576,"threshold_uncertainty_score":0.02694583},"labels":[],"label_agreement":null},{"id":"W2990169293","doi":"10.1109/isncc.2019.8909159","title":"Finding better learning algorithms for self-driving cars: An overview of the LAOP platform","year":2019,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Collège de Maisonneuve; Laboratoire Recherche Informatique Maisonneuve","funders":"","keywords":"Neuroevolution; Computer science; Artificial neural network; Context (archaeology); Artificial intelligence; Process (computing); Machine learning; Deep learning; Java; Algorithm","score_opus":0.06599057292254697,"score_gpt":0.3095288898413565,"score_spread":0.24353831691880956,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2990169293","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015606751,0.0036484867,0.9699149,0.0005226211,0.00008760027,0.00011420413,0.00006590939,0.002449699,0.007589787],"genre_scores_gemma":[0.19671172,0.004503575,0.79018354,0.00030146906,0.00017993437,0.00030696584,0.00041552886,0.0008285408,0.006568749],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99969923,0.000065582186,0.000020454288,0.000068172834,0.00011948183,0.000026957028],"domain_scores_gemma":[0.9996314,0.00016857494,0.00002490109,0.000050034236,0.000085290834,0.0000396884],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00063350995,0.0007628476,0.00046585963,0.00054618303,0.00035103294,0.000992754,0.00132363,0.0007861325,0.004645683],"category_scores_gemma":[0.0012674478,0.00034627886,0.0004660657,0.00047846852,0.0005129082,0.0011965738,0.0010740637,0.0015461382,0.0013350315],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019048637,0.00034810865,0.0012525298,0.00044386284,0.00008517445,0.00011402276,0.0001777292,0.3498711,0.012718888,0.02915462,0.0063524507,0.59929097],"study_design_scores_gemma":[0.00004155957,0.0003574266,0.0004328472,0.000080773905,0.000030894855,0.0001283996,0.000052397474,0.9262524,0.007425121,0.021192594,0.043969348,0.000036142108],"about_ca_topic_score_codex":0.0014705306,"about_ca_topic_score_gemma":0.000989857,"teacher_disagreement_score":0.004645683,"about_ca_system_score_codex":0.0004605925,"about_ca_system_score_gemma":0.0005886893,"threshold_uncertainty_score":0.015541315},"labels":[],"label_agreement":null},{"id":"W2990933479","doi":"10.48550/arxiv.1911.08610","title":"Efficient decorrelation of features using Gramian in Reinforcement Learning","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Decorrelation; Regularization (linguistics); Computer science; Sample complexity; Artificial intelligence; Gramian matrix; Temporal difference learning; Computational complexity theory; Mathematical optimization; Machine learning; Mathematics; Algorithm","score_opus":0.053181348214927066,"score_gpt":0.20262543750929482,"score_spread":0.14944408929436775,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2990933479","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.050327856,0.00014682647,0.9475375,0.00026667054,0.000027417049,0.00005707064,0.000032485415,0.0004668107,0.00113735],"genre_scores_gemma":[0.8746221,0.00009380962,0.12325529,0.00016131249,0.00002876286,0.00014381157,0.00006741965,0.00007790734,0.0015497139],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986285,0.000708038,0.000053047068,0.00025107333,0.00022545336,0.00013383478],"domain_scores_gemma":[0.99643195,0.0024100763,0.00035409885,0.00038729262,0.0002497117,0.00016683686],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002213595,0.0010259433,0.0014724188,0.00044687628,0.00048612957,0.00069398165,0.0011293052,0.001142382,0.0012346879],"category_scores_gemma":[0.0099709,0.0005368267,0.0005357412,0.00043658775,0.002212024,0.0016014408,0.0017090825,0.0019635116,0.0002978451],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002523603,0.00014634535,0.001252667,0.00006928058,0.000048025562,0.00013283521,0.000104944695,0.90101206,0.005026169,0.025984034,0.0010818238,0.06488939],"study_design_scores_gemma":[0.000014274968,0.00004632634,0.00007601668,0.0000040275954,0.000003767111,0.000011902103,0.000004200072,0.98919684,0.0006336817,0.009872011,0.00013026314,0.0000065299496],"about_ca_topic_score_codex":0.002559814,"about_ca_topic_score_gemma":0.0026778206,"teacher_disagreement_score":0.002559814,"about_ca_system_score_codex":0.0009250746,"about_ca_system_score_gemma":0.001200707,"threshold_uncertainty_score":0.0117067695},"labels":[],"label_agreement":null},{"id":"W2991419354","doi":"10.1109/itsc.2019.8916928","title":"Multi-lane Cruising Using Hierarchical Planning and Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada)","funders":"","keywords":"Reinforcement learning; Abstraction; Computer science; Modular design; Hierarchy; Motion (physics); Motion planning; Artificial intelligence; State space; Action (physics); Set (abstract data type); Human–computer interaction; Robot","score_opus":0.03906097747933342,"score_gpt":0.29254376067269544,"score_spread":0.25348278319336204,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2991419354","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044188526,0.000082556464,0.9522822,0.00011074786,0.000016502854,0.000057188245,0.000019649387,0.00061819697,0.0026245136],"genre_scores_gemma":[0.9130532,0.000053400723,0.085393526,0.000037396436,0.000007348599,0.000067934045,0.000039734943,0.000028211247,0.0013190752],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999747,0.00006187338,0.000013570937,0.0000604406,0.00007152019,0.00004558099],"domain_scores_gemma":[0.99954826,0.00016977155,0.00007331641,0.000071675095,0.000075223885,0.00006176576],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005102392,0.0005248567,0.00037789342,0.00023653645,0.00024992094,0.00043498632,0.00096429075,0.00050920894,0.001088837],"category_scores_gemma":[0.0011472983,0.00030433637,0.00043603065,0.00015285707,0.0008260368,0.00058273063,0.0009072112,0.0009256715,0.00018969807],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000040575396,0.000050871884,0.0006889126,0.000027260938,0.000023061248,0.000058534086,0.00006820107,0.96049345,0.004367879,0.005328838,0.00024210093,0.02861024],"study_design_scores_gemma":[0.000005570099,0.000025732308,0.00009430439,0.0000025280642,0.000003819419,0.000006382558,0.0000038671765,0.99672395,0.00057568966,0.002391601,0.00016355817,0.0000030309182],"about_ca_topic_score_codex":0.0067970296,"about_ca_topic_score_gemma":0.007876249,"teacher_disagreement_score":0.0067970296,"about_ca_system_score_codex":0.00072126754,"about_ca_system_score_gemma":0.0010751279,"threshold_uncertainty_score":0.013514876},"labels":[],"label_agreement":null},{"id":"W2991820108","doi":"10.48550/arxiv.1912.04002","title":"Learning Sparse Representations Incrementally in Deep Reinforcement Learning","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Artificial intelligence; Computer science; Machine learning; Cognitive science; Psychology","score_opus":0.0612828309903429,"score_gpt":0.21055536665514085,"score_spread":0.14927253566479795,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2991820108","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15114719,0.00021169662,0.84571046,0.00040022333,0.000047410285,0.000075762146,0.000048755042,0.0007724953,0.0015858785],"genre_scores_gemma":[0.95823014,0.00006369104,0.040717125,0.00010290096,0.000018036468,0.000068170266,0.000048196245,0.000026190453,0.0007255975],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995276,0.00017589962,0.000020249021,0.00007494686,0.000119099685,0.000082199906],"domain_scores_gemma":[0.9976525,0.0015381406,0.00024838586,0.00020325997,0.0002365808,0.000121165285],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013397607,0.000612667,0.0007202445,0.00030330635,0.00021811559,0.00048831914,0.000971877,0.000751711,0.00093859393],"category_scores_gemma":[0.0068003647,0.00043868282,0.00027054801,0.0002610265,0.0010125967,0.0012430731,0.0011042615,0.0016815596,0.00015319078],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012367728,0.00012590739,0.0013133946,0.000047438913,0.000037383852,0.00007708979,0.00006362197,0.9458455,0.0036294993,0.0075987987,0.00060060783,0.040537205],"study_design_scores_gemma":[0.000007652513,0.000023212842,0.000053638814,0.0000019383729,0.0000024650803,0.000005622305,0.0000026320167,0.9962478,0.0004399893,0.0031534228,0.000059498594,0.0000021080525],"about_ca_topic_score_codex":0.002561888,"about_ca_topic_score_gemma":0.003258569,"teacher_disagreement_score":0.002561888,"about_ca_system_score_codex":0.0006260411,"about_ca_system_score_gemma":0.0007812313,"threshold_uncertainty_score":0.0070854425},"labels":[],"label_agreement":null},{"id":"W2995031981","doi":"10.48550/arxiv.1912.05109","title":"Doubly Robust Off-Policy Actor-Critic Algorithms for Reinforcement Learning","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Estimator; Computer science; Variance (accounting); Function (biology); Bellman equation; Value (mathematics); Mathematical optimization; Artificial intelligence; Machine learning; Mathematics; Economics; Statistics","score_opus":0.10240831877352728,"score_gpt":0.2271524172624324,"score_spread":0.12474409848890512,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2995031981","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003067806,0.00021589137,0.99526453,0.000088476074,0.000027715489,0.000022847176,0.000013943986,0.00020372887,0.0010950207],"genre_scores_gemma":[0.71332955,0.00052991515,0.28008103,0.00024764912,0.000104262326,0.00027536482,0.00015116962,0.00028975503,0.0049912175],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983754,0.0007090909,0.00008952006,0.0002645775,0.00043789853,0.0001235774],"domain_scores_gemma":[0.9941702,0.003848028,0.0005519278,0.00052444084,0.00072652777,0.00017891375],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035246778,0.0014443664,0.0018065554,0.00073785917,0.00043320283,0.0013864178,0.0016412844,0.0015159516,0.0020521844],"category_scores_gemma":[0.013831962,0.0007127197,0.000648469,0.0005713018,0.0015813231,0.0012318416,0.0017791291,0.0023189576,0.0005811196],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006490528,0.00003451396,0.00044377494,0.00006799693,0.000042432497,0.00004012649,0.00004247509,0.9371818,0.0008164041,0.023969892,0.00060139224,0.03669429],"study_design_scores_gemma":[0.0000038633125,0.000012696559,0.000026577303,0.00000489552,0.0000028284012,0.0000060708517,0.000001379364,0.9951338,0.00023940022,0.004366127,0.00019865343,0.0000036583563],"about_ca_topic_score_codex":0.0025435367,"about_ca_topic_score_gemma":0.0017168566,"teacher_disagreement_score":0.0035246778,"about_ca_system_score_codex":0.0011413628,"about_ca_system_score_gemma":0.0011496986,"threshold_uncertainty_score":0.018640518},"labels":[],"label_agreement":null},{"id":"W2995040055","doi":"","title":"Reinforcement Learning with Competitive Ensembles of Information-Constrained Primitives","year":2020,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Reinforcement learning; Computer science; Context (archaeology); Generalization; Artificial intelligence; Decomposition; State (computer science); Machine learning; Mathematics","score_opus":0.035859797675240924,"score_gpt":0.28773296694383693,"score_spread":0.251873169268596,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2995040055","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2262359,0.00023409443,0.7651577,0.0006098082,0.00007493919,0.0001411488,0.000070871676,0.0008173859,0.006658184],"genre_scores_gemma":[0.96618974,0.000042679156,0.0321318,0.00009137716,0.000015769661,0.0000784876,0.00004150745,0.00002352323,0.0013849831],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988752,0.00037212085,0.00006275684,0.00023566281,0.0002723764,0.0001819181],"domain_scores_gemma":[0.99698454,0.0014553461,0.00039341985,0.00043739565,0.00036120554,0.00036818287],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020807607,0.00079659827,0.0011494237,0.00037480297,0.0004258653,0.0009927446,0.0016709722,0.0010090504,0.0021912812],"category_scores_gemma":[0.006785824,0.00047883755,0.00050670205,0.00033406226,0.0014625682,0.0016961568,0.0016352097,0.0015536249,0.00031184484],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020356945,0.0001322902,0.0009853044,0.000041651576,0.000049378486,0.000066473265,0.00008453937,0.9475821,0.0028116896,0.02088561,0.00055882527,0.026598603],"study_design_scores_gemma":[0.00001801032,0.00003935879,0.000076488875,0.000002341392,0.0000045800625,0.0000061846745,0.0000046367873,0.9928604,0.00034686876,0.0065162624,0.00012066208,0.0000043315963],"about_ca_topic_score_codex":0.004258293,"about_ca_topic_score_gemma":0.0039788634,"teacher_disagreement_score":0.004258293,"about_ca_system_score_codex":0.0013824366,"about_ca_system_score_gemma":0.0014591612,"threshold_uncertainty_score":0.0110042095},"labels":[],"label_agreement":null},{"id":"W2995372087","doi":"","title":"Learning the Arrow of Time for Problems in Reinforcement Learning","year":2020,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Arrow; Arrow of time; Computer science; Reinforcement learning; Markov decision process; Reachability; Artificial intelligence; Machine learning; Class (philosophy); Function (biology); Markov process; Selection (genetic algorithm); Process (computing); Theoretical computer science; Mathematics","score_opus":0.05955057235187068,"score_gpt":0.3172682931974696,"score_spread":0.2577177208455989,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2995372087","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017267097,0.00052513916,0.9798452,0.0008020037,0.000038062295,0.00002315368,0.00004201003,0.000115416085,0.0013418765],"genre_scores_gemma":[0.6324654,0.0012950257,0.3617703,0.00027528196,0.00018418425,0.00020694984,0.00019830784,0.00012396622,0.0034805734],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99905425,0.0004534036,0.00005860139,0.00023951712,0.00012537376,0.00006882441],"domain_scores_gemma":[0.99496585,0.003940517,0.00041599697,0.0002753604,0.00019657551,0.00020580433],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025535454,0.00082421815,0.0008684381,0.00061711064,0.0006274078,0.0017808591,0.0011012885,0.0016217922,0.0035505889],"category_scores_gemma":[0.013876712,0.00041205937,0.0008505912,0.00062356505,0.0027047666,0.005086039,0.0018074982,0.00458183,0.00033410208],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017996978,0.00009638528,0.0014966726,0.00023187054,0.00006269703,0.000071768045,0.0002536391,0.47762123,0.0018029877,0.46125603,0.0014264098,0.055500425],"study_design_scores_gemma":[0.0000148126055,0.000039565315,0.00013896807,0.00002447399,0.0000069241732,0.000012455372,0.000023380922,0.6537325,0.00038823427,0.3446155,0.0009920171,0.000011225374],"about_ca_topic_score_codex":0.0024104628,"about_ca_topic_score_gemma":0.0022557194,"teacher_disagreement_score":0.0035505889,"about_ca_system_score_codex":0.0018671615,"about_ca_system_score_gemma":0.0011765256,"threshold_uncertainty_score":0.013547242},"labels":[],"label_agreement":null},{"id":"W2995509794","doi":"10.48550/arxiv.2002.06487","title":"Maxmin Q-learning: Controlling the Estimation Bias of Q-learning","year":2020,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Convergence (economics); Benchmark (surveying); Computer science; Variance (accounting); Generalization error; Generalization; Artificial intelligence; Inductive bias; Machine learning; Q-learning; Algorithm; Value (mathematics); Mathematics; Stability (learning theory); Multi-task learning; Reinforcement learning","score_opus":0.09264049702283787,"score_gpt":0.1833738516716606,"score_spread":0.09073335464882272,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2995509794","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010990199,0.0002491342,0.9874084,0.00024918103,0.00003691139,0.000041683885,0.000018867637,0.00035467136,0.0006510157],"genre_scores_gemma":[0.6614153,0.00029233424,0.3355082,0.00069836946,0.00013130574,0.0002720601,0.00009430219,0.00023782179,0.0013504255],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9945462,0.0031821497,0.00024218422,0.0009977765,0.00069300085,0.0003387054],"domain_scores_gemma":[0.97352207,0.019646993,0.0018704619,0.002509147,0.0019701847,0.00048112738],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010857264,0.0014113339,0.0020182887,0.0006558426,0.0006666941,0.0015962844,0.003048555,0.0019101814,0.0017642591],"category_scores_gemma":[0.04126065,0.0007014133,0.00058748224,0.00085793703,0.0026776614,0.002801091,0.0025677704,0.0027661934,0.00040408643],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004054018,0.0001955934,0.004026164,0.00031152455,0.00017146928,0.00009114876,0.00021152581,0.7800384,0.0030947914,0.049526874,0.0027786675,0.15914844],"study_design_scores_gemma":[0.00004018464,0.00008117637,0.00019547643,0.000022475662,0.000012730288,0.000028277984,0.000010651984,0.97479874,0.0015411328,0.02275782,0.00050048594,0.000010920834],"about_ca_topic_score_codex":0.0022144364,"about_ca_topic_score_gemma":0.0016594975,"teacher_disagreement_score":0.010857264,"about_ca_system_score_codex":0.0014615782,"about_ca_system_score_gemma":0.002630563,"threshold_uncertainty_score":0.05741942},"labels":[],"label_agreement":null},{"id":"W2996001434","doi":"10.48550/arxiv.1912.05128","title":"Marginalized State Distribution Entropy Regularization in Policy Optimization","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Regularization (linguistics); Entropy (arrow of time); Mathematical optimization; Econometrics; Mathematics; Economics; Computer science; Statistical physics; Mathematical economics; Political science; Physics; Artificial intelligence; Thermodynamics","score_opus":0.03583607267643475,"score_gpt":0.1904179922911267,"score_spread":0.15458191961469192,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2996001434","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011137886,0.000318556,0.98658293,0.00027894575,0.000038967166,0.000027277427,0.00003063954,0.00020420198,0.0013804722],"genre_scores_gemma":[0.8629543,0.0005522032,0.13204767,0.0003344502,0.00015357422,0.0002222925,0.0001625562,0.00020925725,0.0033636496],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99843544,0.00080870907,0.0000696953,0.00023961313,0.0003229505,0.00012360887],"domain_scores_gemma":[0.9962321,0.0028023447,0.00023862947,0.00029498606,0.00030727393,0.00012473592],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024951797,0.0010318732,0.001376798,0.0006674499,0.000442923,0.0011532552,0.001013924,0.0013298696,0.0019477375],"category_scores_gemma":[0.01091154,0.00060210755,0.0007156251,0.00073911034,0.0019468182,0.0020027813,0.0017428185,0.0022890368,0.00036872196],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000099312834,0.00006633869,0.0006866781,0.00008396856,0.0000571529,0.000049724964,0.00006876883,0.9212264,0.0019357143,0.043920975,0.0009943589,0.030810654],"study_design_scores_gemma":[0.0000073362594,0.000018380582,0.00007614457,0.000007348843,0.0000040798413,0.000007938231,0.0000028987133,0.9804863,0.00045846115,0.018688986,0.00023723839,0.0000049466653],"about_ca_topic_score_codex":0.0023653496,"about_ca_topic_score_gemma":0.0017105158,"teacher_disagreement_score":0.0024951797,"about_ca_system_score_codex":0.0014868135,"about_ca_system_score_gemma":0.0013773948,"threshold_uncertainty_score":0.013195932},"labels":[],"label_agreement":null},{"id":"W2996026485","doi":"10.48550/arxiv.1912.06875","title":"Natural Actor-Critic Converges Globally for Hierarchical Linear Quadratic Regulator","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Booth University College","funders":"","keywords":"Curse of dimensionality; Reinforcement learning; Linear-quadratic regulator; Convergence (economics); Mathematical optimization; State space; Quadratic equation; Computer science; Action (physics); Mathematics; Optimal control; Artificial intelligence","score_opus":0.05382236452644264,"score_gpt":0.20893269833253675,"score_spread":0.15511033380609413,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2996026485","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016096327,0.0002214532,0.9768287,0.00040323433,0.00003632673,0.000043635257,0.00003749789,0.00032540335,0.0060074576],"genre_scores_gemma":[0.8820672,0.0002419443,0.11046794,0.0002754293,0.00005050698,0.00027308764,0.00013258237,0.00014658243,0.0063447985],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99935216,0.00022806821,0.000028438237,0.00016729564,0.00014740294,0.00007668993],"domain_scores_gemma":[0.99740064,0.0018182878,0.00023611405,0.00013006097,0.00030062036,0.000114298],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019078583,0.0009470129,0.0011516052,0.00042015265,0.00042759944,0.0007696291,0.0008971244,0.0011941202,0.0026533345],"category_scores_gemma":[0.006394231,0.00038301636,0.0005278443,0.00033180616,0.0013511524,0.0008551401,0.0015764739,0.0015875108,0.0004675118],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000036462174,0.00003146361,0.00039995465,0.00009631731,0.000027870507,0.000062487765,0.00008581888,0.9355814,0.0012881847,0.0498046,0.0012212634,0.011364212],"study_design_scores_gemma":[0.000005029638,0.00001133374,0.000031173447,0.0000035094786,0.0000017113297,0.000004676657,0.000003924987,0.9903673,0.0000839285,0.009320157,0.00016527424,0.000001984108],"about_ca_topic_score_codex":0.004227535,"about_ca_topic_score_gemma":0.0036070084,"teacher_disagreement_score":0.004227535,"about_ca_system_score_codex":0.0012412523,"about_ca_system_score_gemma":0.0013974946,"threshold_uncertainty_score":0.010089815},"labels":[],"label_agreement":null},{"id":"W2996283994","doi":"10.48550/arxiv.2002.05822","title":"Frequency-based Search-control in Dyna","year":2020,"lang":"en","type":"article","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Reinforcement learning; Bellman equation; Benchmark (surveying); Artificial intelligence; Function (biology); Machine learning; Mathematical optimization; Mathematics","score_opus":0.014531650295161206,"score_gpt":0.23006169952479144,"score_spread":0.21553004922963023,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2996283994","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0418553,0.00027457587,0.9528864,0.00028823642,0.00004371944,0.00007187722,0.000046892572,0.0006947014,0.0038382201],"genre_scores_gemma":[0.89053595,0.00011666398,0.106706046,0.00014027701,0.000021843547,0.0001876543,0.000049847975,0.00009612555,0.0021455823],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991291,0.00031975325,0.000059770282,0.00020281518,0.00020218118,0.000086428154],"domain_scores_gemma":[0.99658597,0.0023031163,0.0003725015,0.000274934,0.0002955192,0.00016809856],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018729987,0.0008148895,0.00095103524,0.0006492015,0.0005994793,0.0010925885,0.0011158622,0.00090326904,0.0029026957],"category_scores_gemma":[0.007565818,0.00046353054,0.00046160168,0.00039610744,0.0018020587,0.0013509092,0.0014187449,0.0012879868,0.0002998138],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016095095,0.00008861767,0.0014911921,0.00014469879,0.000059395537,0.00008306754,0.00017037576,0.86575925,0.004349256,0.08016598,0.0009504954,0.04657681],"study_design_scores_gemma":[0.000017831002,0.000045858334,0.00009400128,0.000008200223,0.0000070056067,0.000016544605,0.0000081367825,0.9822218,0.0007647139,0.016342618,0.00046464632,0.000008673578],"about_ca_topic_score_codex":0.004240556,"about_ca_topic_score_gemma":0.00411582,"teacher_disagreement_score":0.004240556,"about_ca_system_score_codex":0.0013173373,"about_ca_system_score_gemma":0.0014210267,"threshold_uncertainty_score":0.009905517},"labels":[],"label_agreement":null},{"id":"W2996347495","doi":"","title":"Exploring Model-based Planning with Policy Networks","year":2020,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Benchmarking; Artificial neural network; Code (set theory); Artificial intelligence; Mathematical optimization; Sample (material); Optimization problem; Control (management); Action (physics); State space; Machine learning; Algorithm; Mathematics","score_opus":0.22710027386392856,"score_gpt":0.1933980817790889,"score_spread":0.033702192084839655,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2996347495","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03994743,0.00028958157,0.9557342,0.0003145888,0.00003168009,0.00004644298,0.00005955014,0.00082862726,0.0027479818],"genre_scores_gemma":[0.81750417,0.00029214055,0.1794038,0.00018668504,0.00003113269,0.00026371967,0.00018122802,0.00014788382,0.0019893057],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995833,0.00017442834,0.00001900826,0.00009209206,0.00008247737,0.000048630336],"domain_scores_gemma":[0.9987085,0.0010045394,0.00008075763,0.00009050672,0.000077167424,0.000038562426],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00092699635,0.0010629856,0.0009235229,0.000488079,0.0003706635,0.000851089,0.0009581726,0.0010094374,0.002099963],"category_scores_gemma":[0.0034464824,0.0007407936,0.00060965325,0.00043193073,0.0012028196,0.0014272641,0.0011968241,0.0013629262,0.00027860943],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002531414,0.00002186346,0.00018264443,0.000024763303,0.0000120371515,0.000020035352,0.000019177543,0.9814368,0.00036442024,0.0057513653,0.00023185616,0.011909669],"study_design_scores_gemma":[0.000004908865,0.0000070355077,0.000012374071,0.0000020489535,0.0000016130529,0.000002173011,0.0000021083288,0.9964239,0.00011022843,0.0033132962,0.00011915978,0.0000011769057],"about_ca_topic_score_codex":0.007857161,"about_ca_topic_score_gemma":0.0071049035,"teacher_disagreement_score":0.007857161,"about_ca_system_score_codex":0.0011517364,"about_ca_system_score_gemma":0.0016309042,"threshold_uncertainty_score":0.015622854},"labels":[],"label_agreement":null},{"id":"W2997289589","doi":"10.1609/aaai.v34i04.5955","title":"Count-Based Exploration with the Successor Representation","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":106,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Alberta Innovates; University of Alberta; Alberta Machine Intelligence Institute; Compute Canada","keywords":"Successor cardinal; Representation (politics); Norm (philosophy); Computer science; Sample complexity; Generalization; Reinforcement learning; Similarity (geometry); Artificial intelligence; State (computer science); Algorithm; Theoretical computer science; Mathematics","score_opus":0.05162394606866699,"score_gpt":0.263603992355036,"score_spread":0.21198004628636902,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2997289589","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02740588,0.00020498481,0.9676494,0.0003616572,0.000053353724,0.00006878617,0.0000974993,0.00087398436,0.0032844474],"genre_scores_gemma":[0.6898327,0.00015381849,0.30387384,0.0002174671,0.000057815487,0.00031279863,0.0002552164,0.00018980539,0.0051065874],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99907,0.0003261522,0.000060862483,0.00020814103,0.00023757262,0.00009725537],"domain_scores_gemma":[0.9969981,0.0017876491,0.00028567805,0.0004745996,0.0002592752,0.00019467695],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016828057,0.0007093452,0.0012010874,0.0007489717,0.0004694233,0.0012070776,0.0021744943,0.0012194883,0.0047348724],"category_scores_gemma":[0.008043453,0.00037879674,0.00062377757,0.000742825,0.0014032051,0.0036730815,0.0025938659,0.0019167501,0.0006386793],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003936095,0.0001718341,0.0019184413,0.00019957813,0.000063709485,0.00012326366,0.0002134152,0.54964614,0.0038685913,0.22493726,0.0033274668,0.21513669],"study_design_scores_gemma":[0.000024075425,0.000067714405,0.00007536897,0.000012790704,0.0000072188477,0.000027687242,0.000008651916,0.94101137,0.00079063437,0.057228047,0.00073656504,0.000009859263],"about_ca_topic_score_codex":0.0012062312,"about_ca_topic_score_gemma":0.0017337623,"teacher_disagreement_score":0.0047348724,"about_ca_system_score_codex":0.0010047924,"about_ca_system_score_gemma":0.0016694186,"threshold_uncertainty_score":0.015839756},"labels":[],"label_agreement":null},{"id":"W2998461398","doi":"10.1609/aaai.v34i04.5784","title":"Fixed-Horizon Temporal Difference Methods for Stable Reinforcement Learning","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Huawei Technologies (Canada); University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates; DeepMind","keywords":"Reinforcement learning; Bellman equation; Horizon; Temporal difference learning; Function (biology); Time horizon; Value (mathematics); Stability (learning theory); Computer science; Function approximation; Mathematics; Mathematical optimization; Artificial intelligence; Machine learning; Artificial neural network","score_opus":0.05269488141129239,"score_gpt":0.32804208255578104,"score_spread":0.27534720114448863,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2998461398","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002498533,0.00019771978,0.9958574,0.00009258442,0.000040876133,0.00001809291,0.000013428913,0.00008956917,0.0011918702],"genre_scores_gemma":[0.54561985,0.00056029984,0.44769222,0.00028100846,0.0000886076,0.00032178045,0.00009837484,0.00018681739,0.005151144],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99942505,0.00023070413,0.000028414908,0.00010293285,0.0001644997,0.000048446007],"domain_scores_gemma":[0.9973502,0.0019320458,0.00019815464,0.00014189836,0.00028097833,0.0000966486],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001892933,0.0008204663,0.0008838018,0.00037369443,0.00037376818,0.000878088,0.0017420714,0.0010393932,0.0039119967],"category_scores_gemma":[0.006435315,0.00039371217,0.0006156992,0.00041340635,0.0012649538,0.0013039564,0.0012493999,0.0021600577,0.0004993501],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008757655,0.0000663014,0.00048074083,0.00015047872,0.000047522088,0.000056520807,0.00008397981,0.8471003,0.001722734,0.10003822,0.0010926747,0.049073],"study_design_scores_gemma":[0.000008108379,0.000016799046,0.000019261743,0.000008066338,0.000003041904,0.0000060719112,0.000002645868,0.98438466,0.00026912257,0.014756307,0.0005224794,0.0000035187034],"about_ca_topic_score_codex":0.0034977184,"about_ca_topic_score_gemma":0.0025130466,"teacher_disagreement_score":0.0039119967,"about_ca_system_score_codex":0.0012530062,"about_ca_system_score_gemma":0.00112703,"threshold_uncertainty_score":0.013086975},"labels":[],"label_agreement":null},{"id":"W2998494185","doi":"10.1609/aaai.v34i04.6027","title":"Gamma-Nets: Generalizing Value Estimation over Timescale","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates; Alberta Machine Intelligence Institute; Compute Canada","keywords":"Reinforcement learning; Computer science; Estimator; Bellman equation; Abstraction; Artificial intelligence; Scalability; Function (biology); Set (abstract data type); Temporal difference learning; Machine learning; Value (mathematics); Representation (politics); Key (lock); Markov decision process; Mathematical optimization; Markov process; Mathematics; Statistics","score_opus":0.020430689637633725,"score_gpt":0.2437083033784362,"score_spread":0.22327761374080246,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2998494185","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022689054,0.0003960851,0.9733612,0.00040187596,0.000072147486,0.00004923847,0.00017566793,0.0012044,0.0016503222],"genre_scores_gemma":[0.74909496,0.0006920793,0.2440497,0.00054960424,0.00012787744,0.000249509,0.00062069105,0.0003590758,0.0042565414],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99939966,0.00015749187,0.000039053455,0.00018656183,0.00013508766,0.00008211402],"domain_scores_gemma":[0.99775106,0.0013725082,0.00026685058,0.0002253468,0.00025335455,0.0001308166],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021491363,0.0015829562,0.0011112954,0.00079645164,0.00046156315,0.0012660252,0.0024697199,0.0015935161,0.00231283],"category_scores_gemma":[0.007778883,0.0008605263,0.0010128287,0.00060570205,0.0014525979,0.0034730448,0.0020500908,0.0033892775,0.00048357103],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008299572,0.00003008702,0.0012644123,0.00004460598,0.000035885758,0.00005779808,0.00006400648,0.9384231,0.0009370685,0.012917353,0.0011489829,0.044993788],"study_design_scores_gemma":[0.0000050533995,0.000012197263,0.000059376413,0.0000069420203,0.000004220977,0.000006215806,0.0000044295325,0.9863127,0.00029453577,0.0130227525,0.00026770207,0.000004013421],"about_ca_topic_score_codex":0.013551806,"about_ca_topic_score_gemma":0.012117948,"teacher_disagreement_score":0.013551806,"about_ca_system_score_codex":0.0021609932,"about_ca_system_score_gemma":0.0016301576,"threshold_uncertainty_score":0.02694583},"labels":[],"label_agreement":null},{"id":"W3000091990","doi":"10.1038/s41593-019-0574-1","title":"Causal evidence supporting the proposal that dopamine transients function as temporal difference prediction errors","year":2020,"lang":"en","type":"article","venue":"Nature Neuroscience","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":103,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada; U.S. Department of Health and Human Services; Government of Canada; National Institutes of Health; Concordia University; National Institute on Drug Abuse; Canada Research Chairs","keywords":"Neuroscience; Dopamine; Temporal difference learning; Psychology; Function (biology); Cognitive psychology; Biology; Computer science; Artificial intelligence; Reinforcement learning","score_opus":0.03897630391855688,"score_gpt":0.28451362114193823,"score_spread":0.24553731722338135,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3000091990","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22715548,0.00087480457,0.71324116,0.00864622,0.0010832287,0.00008732471,0.000652132,0.0007176737,0.04754203],"genre_scores_gemma":[0.9759814,0.0002973495,0.020902444,0.00041339028,0.00011862447,0.000024133466,0.00015344517,0.000048880916,0.0020604234],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.999288,0.00013511456,0.000053483633,0.00023257593,0.00023109415,0.00005975796],"domain_scores_gemma":[0.9834187,0.009973265,0.0021552525,0.0022755798,0.001573345,0.00060389005],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021700903,0.00042481592,0.00039315244,0.000756497,0.00043206787,0.0013460021,0.0013571784,0.0018356529,0.008678238],"category_scores_gemma":[0.021715738,0.00042767668,0.00059441844,0.0005423176,0.0018228403,0.0022436343,0.0009504579,0.0020119702,0.00079427427],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00093307893,0.00038229683,0.020273479,0.00038268624,0.00027268537,0.00086622994,0.00028485636,0.04146288,0.032855622,0.8160981,0.003468312,0.08271973],"study_design_scores_gemma":[0.00016304283,0.000209075,0.016239287,0.00005137759,0.00007125844,0.0006729389,0.0000863772,0.18678159,0.009110383,0.7823688,0.0041718925,0.000074058626],"about_ca_topic_score_codex":0.0014637009,"about_ca_topic_score_gemma":0.0011687801,"teacher_disagreement_score":0.008678238,"about_ca_system_score_codex":0.0006445506,"about_ca_system_score_gemma":0.00062786153,"threshold_uncertainty_score":0.029031575},"labels":[],"label_agreement":null},{"id":"W3000711993","doi":"10.1109/ro-man46459.2019.8956320","title":"Incremental Estimation of Users’ Expertise Level","year":2019,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Estimation; Engineering; Systems engineering","score_opus":0.03310113847019799,"score_gpt":0.2636222543227014,"score_spread":0.23052111585250343,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3000711993","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.560768,0.00034678346,0.4344592,0.0001968994,0.000023079101,0.000117358686,0.0002559873,0.0011912029,0.0026415542],"genre_scores_gemma":[0.9640794,0.000042368036,0.035427514,0.000017989834,0.000009417171,0.000034345678,0.00010564272,0.000017792969,0.0002654308],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988539,0.00028994086,0.00007904839,0.00037982565,0.00029160499,0.00010568315],"domain_scores_gemma":[0.98879856,0.0061446005,0.0012765218,0.0011114067,0.002247436,0.0004214383],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014548531,0.0007266028,0.0006781329,0.0010846241,0.000208184,0.000687664,0.0008591871,0.00068709475,0.00087981333],"category_scores_gemma":[0.018445095,0.00038461864,0.0002886824,0.00036428362,0.00033841387,0.0014413313,0.0011026296,0.0007614749,0.00036343673],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012282364,0.000547768,0.20272806,0.00044183864,0.00029792014,0.00035476367,0.0017845834,0.20918746,0.045316994,0.0017384383,0.0021218532,0.53425205],"study_design_scores_gemma":[0.000028445682,0.00044796366,0.047532078,0.00003431328,0.000059569204,0.00024167901,0.00024118008,0.93494326,0.012883765,0.0027218894,0.0007979428,0.00006797511],"about_ca_topic_score_codex":0.0039774217,"about_ca_topic_score_gemma":0.004636526,"teacher_disagreement_score":0.0039774217,"about_ca_system_score_codex":0.00042720995,"about_ca_system_score_gemma":0.0005939671,"threshold_uncertainty_score":0.007908523},"labels":[],"label_agreement":null},{"id":"W3004681481","doi":"10.65109/lvmd6651","title":"Multi Type Mean Field Reinforcement Learning","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Collège Boréal; University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Field (mathematics); Mean field theory; Type (biology); Core (optical fiber); Artificial intelligence; Game theory; Relaxation (psychology); Multi-agent system; Machine learning; Mathematics; Mathematical economics","score_opus":0.055785596669593734,"score_gpt":0.29388675866132424,"score_spread":0.2381011619917305,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3004681481","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04395012,0.00019143095,0.9519728,0.00028868817,0.00009084756,0.000067261535,0.00003478309,0.0005513876,0.0028526923],"genre_scores_gemma":[0.8621253,0.000119733035,0.13353416,0.00025010185,0.000053998177,0.00013916189,0.000075957425,0.000067737026,0.003633882],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99923587,0.00031997476,0.0000315571,0.00015181437,0.00017320226,0.00008753063],"domain_scores_gemma":[0.997297,0.0015713556,0.00024019372,0.0003466608,0.00034257144,0.00020230314],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001972732,0.00076574495,0.001189378,0.0004532505,0.00045090943,0.00083693804,0.0016465441,0.0011343181,0.0018130514],"category_scores_gemma":[0.005708608,0.00030136274,0.00059519516,0.00037187134,0.00115556,0.0013675824,0.001163934,0.0014675229,0.00031385635],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013447119,0.000098229815,0.0008697814,0.00005409474,0.00006025802,0.000059493548,0.000048932925,0.9181123,0.0019106306,0.033058353,0.001671835,0.043921556],"study_design_scores_gemma":[0.000011295862,0.000022722874,0.000047691436,0.0000019437248,0.0000026333453,0.000008180913,0.000002541597,0.9917991,0.00024835952,0.007610982,0.00024163947,0.0000029406326],"about_ca_topic_score_codex":0.0020940485,"about_ca_topic_score_gemma":0.0019707866,"teacher_disagreement_score":0.0020940485,"about_ca_system_score_codex":0.0010480154,"about_ca_system_score_gemma":0.00084449473,"threshold_uncertainty_score":0.010432959},"labels":[],"label_agreement":null},{"id":"W3008080228","doi":"10.1007/978-981-15-1773-0_17","title":"Revisited: Machine Intelligence in Heterogeneous Multi-Agent Systems","year":2020,"lang":"en","type":"book-chapter","venue":"Lecture notes in electrical engineering","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Artificial intelligence","score_opus":0.014693134983030414,"score_gpt":0.22612211726262396,"score_spread":0.21142898227959356,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3008080228","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010506288,0.1551243,0.7354703,0.02117732,0.008564158,0.00005808667,0.00014229871,0.00038943323,0.0685678],"genre_scores_gemma":[0.5598558,0.05826315,0.24324346,0.006718637,0.013297651,0.0002329122,0.00032030593,0.00054893,0.11751914],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9996202,0.00016149526,0.000016560283,0.00007972645,0.000093322895,0.000028669818],"domain_scores_gemma":[0.9993401,0.0004491032,0.000042846794,0.00007601896,0.000055737983,0.00003605613],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00078353065,0.0008046876,0.0011195233,0.00047954195,0.00035265222,0.0029531773,0.0018473617,0.002551711,0.005506007],"category_scores_gemma":[0.0021336065,0.00043620775,0.0006387761,0.0010846794,0.0015356994,0.0032628789,0.0013923735,0.0033954359,0.0010257366],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000057640624,0.000040114268,0.0002112506,0.0006185378,0.00012973866,0.00027984023,0.00016074149,0.061870363,0.0013684961,0.8047503,0.04082087,0.08969211],"study_design_scores_gemma":[0.000028198085,0.00003019474,0.00025295158,0.00014306819,0.000037377667,0.00018242854,0.000056344637,0.17084017,0.0005935992,0.761294,0.06652104,0.00002064994],"about_ca_topic_score_codex":0.0013193897,"about_ca_topic_score_gemma":0.0015066621,"teacher_disagreement_score":0.005506007,"about_ca_system_score_codex":0.0008531079,"about_ca_system_score_gemma":0.000522541,"threshold_uncertainty_score":0.018419445},"labels":[],"label_agreement":null},{"id":"W3009186063","doi":"","title":"Spatially-Distributed Interactive Behaviour Generation for Architecture-Scale Systems Based on Reinforcement Learning","year":2020,"lang":"en","type":"dissertation","venue":"UWSpace (University of Waterloo)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Reinforcement learning; Architecture; Scale (ratio); Computer science; Human–computer interaction; Distributed computing; Data science; Artificial intelligence; Geography; Cartography","score_opus":0.014137990150256935,"score_gpt":0.21692191390495366,"score_spread":0.20278392375469673,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3009186063","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11060912,0.00010897131,0.8849663,0.00013841424,0.000028098764,0.00013290072,0.000020099826,0.00090587523,0.0030901113],"genre_scores_gemma":[0.93616074,0.000041419396,0.062145848,0.000037917558,0.000008031447,0.00013918985,0.000023717295,0.00003917936,0.001403896],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99971277,0.00009519458,0.000015695417,0.00007053121,0.000059060327,0.000046640962],"domain_scores_gemma":[0.99901617,0.000558054,0.00012035548,0.00010497291,0.00012530634,0.000075065414],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007629094,0.00046408406,0.00044979114,0.0002409944,0.00036086823,0.0004802178,0.0010797076,0.00043089726,0.0020584438],"category_scores_gemma":[0.002232624,0.00030530788,0.0004813906,0.0001206363,0.00076361495,0.0004737938,0.0010320948,0.00076261465,0.00026421138],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000067766196,0.00009442653,0.0009918675,0.000038667513,0.000023878727,0.00004968815,0.00013912201,0.9573626,0.0055809184,0.0029696834,0.00023723007,0.032444168],"study_design_scores_gemma":[0.000008081004,0.000032190874,0.00009549445,0.0000018472896,0.000002309537,0.000005623171,0.000005861119,0.9986303,0.0004538682,0.00062202814,0.0001402274,0.0000022385996],"about_ca_topic_score_codex":0.00524031,"about_ca_topic_score_gemma":0.0047917683,"teacher_disagreement_score":0.00524031,"about_ca_system_score_codex":0.00084822095,"about_ca_system_score_gemma":0.00061311154,"threshold_uncertainty_score":0.010419667},"labels":[],"label_agreement":null},{"id":"W3010862467","doi":"10.48550/arxiv.2003.07417","title":"Improving Performance in Reinforcement Learning by Breaking Generalization in Neural Networks","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Generalization; Artificial neural network; Scalability; Machine learning","score_opus":0.03919345243744466,"score_gpt":0.17974001139860277,"score_spread":0.1405465589611581,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3010862467","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13524732,0.0008179937,0.8562963,0.00068504113,0.000116351905,0.00011872949,0.000058790498,0.0023512084,0.0043082787],"genre_scores_gemma":[0.9335115,0.0001800123,0.064564385,0.00028331607,0.000049525577,0.00013019465,0.00006364513,0.00013884416,0.0010785926],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981255,0.00056434516,0.0001414812,0.0004847386,0.00043410412,0.00024986008],"domain_scores_gemma":[0.99222386,0.004801912,0.0007284813,0.0013885275,0.0006254258,0.00023175604],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043418603,0.0016411047,0.0014032694,0.00042778594,0.0005835991,0.00095146307,0.0017709461,0.0013936907,0.0018674006],"category_scores_gemma":[0.020600963,0.000682104,0.00068132434,0.00033934708,0.0021072438,0.0021317068,0.0020908306,0.0035975762,0.00043893466],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022721838,0.00020940324,0.002271263,0.00012560726,0.00010162338,0.0000904911,0.00011829894,0.8884044,0.0070644016,0.007972267,0.0010601878,0.092354834],"study_design_scores_gemma":[0.00002057063,0.00012417002,0.0002637627,0.000011487909,0.00001120884,0.000018445424,0.000007242031,0.9918251,0.001690863,0.0057744216,0.0002452703,0.000007488397],"about_ca_topic_score_codex":0.004941412,"about_ca_topic_score_gemma":0.0032070044,"teacher_disagreement_score":0.004941412,"about_ca_system_score_codex":0.0014908953,"about_ca_system_score_gemma":0.0013488143,"threshold_uncertainty_score":0.022962213},"labels":[],"label_agreement":null},{"id":"W3011813374","doi":"10.1109/cdc40024.2019.9029898","title":"Approximate information state for partially observed systems","year":2019,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Observable; Computer science; State (computer science); Reinforcement learning; Partially observable Markov decision process; Markov decision process; Benchmark (surveying); Markov process; Mathematical optimization; Constructive; Artificial intelligence; Theoretical computer science; Markov chain; Markov model; Machine learning; Mathematics; Algorithm","score_opus":0.026192144294298572,"score_gpt":0.2265800874326913,"score_spread":0.20038794313839275,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3011813374","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013043502,0.00014113433,0.984928,0.00014861602,0.0000100794,0.00003299567,0.00012502317,0.00022604232,0.0013446271],"genre_scores_gemma":[0.7709622,0.00044936186,0.22511348,0.000102622296,0.000041771462,0.0003645692,0.0006670876,0.00014239136,0.0021565268],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9986168,0.00052501267,0.00006866292,0.0002630081,0.00040003753,0.00012646278],"domain_scores_gemma":[0.9948991,0.0037177207,0.000549081,0.00034282234,0.00037097622,0.000120226505],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017050001,0.00088054163,0.0010512412,0.00089829793,0.00039600153,0.0013918006,0.0011384344,0.0010902669,0.0025949923],"category_scores_gemma":[0.009307353,0.0005437015,0.00092811376,0.0007847968,0.0018807164,0.0031966723,0.0013800524,0.0021550786,0.00027827176],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006279682,0.000029829183,0.00046711921,0.00008410142,0.000020833999,0.000049959297,0.000088968525,0.89435405,0.00065682566,0.09212135,0.0003943565,0.011669787],"study_design_scores_gemma":[0.000004665426,0.000014892914,0.000068010726,0.000008631409,0.0000034163272,0.000006375829,0.000006981187,0.96500546,0.00030246435,0.034272853,0.00030181865,0.0000045114357],"about_ca_topic_score_codex":0.0040359357,"about_ca_topic_score_gemma":0.002920592,"teacher_disagreement_score":0.0040359357,"about_ca_system_score_codex":0.0017764593,"about_ca_system_score_gemma":0.0015810174,"threshold_uncertainty_score":0.012889147},"labels":[],"label_agreement":null},{"id":"W3012442074","doi":"10.1109/cdc40024.2019.9029788","title":"Q-Learning with Side Information in Multi-Agent Finite Games","year":2019,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Human–computer interaction","score_opus":0.013505554646719783,"score_gpt":0.22893920249844577,"score_spread":0.21543364785172597,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3012442074","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022628367,0.00018645734,0.9755028,0.00017745336,0.00002879826,0.00003742575,0.000012364144,0.00007894679,0.00134748],"genre_scores_gemma":[0.9198246,0.00017372797,0.0778687,0.00014584964,0.000035348196,0.00014602653,0.000027470194,0.00002531286,0.0017530187],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99874926,0.0006970268,0.000049360675,0.00017147708,0.0002238117,0.00010896004],"domain_scores_gemma":[0.9950335,0.003837154,0.0003095713,0.00023113033,0.0003844984,0.00020405934],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00314332,0.0007963625,0.0015571134,0.00033587168,0.00041795676,0.0007706599,0.0015460227,0.0011522032,0.0011335118],"category_scores_gemma":[0.007407132,0.00046124894,0.00045137238,0.0003312963,0.0018807255,0.001571588,0.0014693093,0.0014971544,0.00019065512],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007317688,0.000046222423,0.0003453091,0.000058910857,0.000030722782,0.000074754076,0.000053725107,0.957153,0.0007127117,0.029201621,0.00021167104,0.012038176],"study_design_scores_gemma":[0.000010437281,0.00001991237,0.000019400175,0.0000028219804,0.0000023353978,0.0000055663854,0.000002181088,0.9909402,0.00011076991,0.008804475,0.00007942693,0.0000024427934],"about_ca_topic_score_codex":0.0025111446,"about_ca_topic_score_gemma":0.0016588956,"teacher_disagreement_score":0.00314332,"about_ca_system_score_codex":0.0009728205,"about_ca_system_score_gemma":0.0013001882,"threshold_uncertainty_score":0.016623676},"labels":[],"label_agreement":null},{"id":"W3013464227","doi":"10.1109/ieeeconf44664.2019.9049021","title":"Decentralized Q-Learning with Constant Aspirations in Stochastic Games","year":2019,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Reinforcement learning; Computer science; Convergence (economics); Constant (computer programming); Mathematical optimization; State (computer science); Control (management); Information structure; Fictitious play; Complete information; Function (biology); Obstacle; Nash equilibrium; Artificial intelligence; Mathematical economics; Mathematics; Algorithm; Economics","score_opus":0.012778840513520693,"score_gpt":0.22521369905343583,"score_spread":0.21243485853991514,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3013464227","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041431848,0.00016608575,0.9545221,0.00041708542,0.000028733892,0.000083251856,0.000036666836,0.00021516309,0.0030990783],"genre_scores_gemma":[0.92462945,0.00012016823,0.07243079,0.0001666844,0.00003257662,0.00016944481,0.000048818303,0.000043949152,0.0023581453],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99860317,0.0005689602,0.000065266206,0.00030097677,0.00024803547,0.00021361288],"domain_scores_gemma":[0.9943462,0.0041247625,0.0005056055,0.00024739347,0.00042876328,0.00034738472],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026083891,0.0008370266,0.0015156117,0.00047762186,0.0006989191,0.0011121194,0.0015277009,0.0012502618,0.002099827],"category_scores_gemma":[0.012101234,0.00059746037,0.0004335974,0.00042084092,0.0022593301,0.0017991902,0.002141346,0.0018370172,0.00027648616],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009085602,0.00006008779,0.00068373984,0.000061492785,0.00002503609,0.00008668756,0.00013303588,0.9251592,0.0007844126,0.057412643,0.00059340167,0.0149093345],"study_design_scores_gemma":[0.000022909417,0.000020264202,0.000054986347,0.000004043228,0.0000024370615,0.000008633799,0.000008328286,0.97697234,0.00011046499,0.022641435,0.00014978532,0.0000043239634],"about_ca_topic_score_codex":0.006563062,"about_ca_topic_score_gemma":0.0049838894,"teacher_disagreement_score":0.006563062,"about_ca_system_score_codex":0.0016279055,"about_ca_system_score_gemma":0.0021930775,"threshold_uncertainty_score":0.013794661},"labels":[],"label_agreement":null},{"id":"W3014454071","doi":"10.48550/arxiv.2004.00993","title":"Augmented Q Imitation Learning (AQIL)","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Reinforcement learning; Imitation; Artificial intelligence; Q-learning; Computer science; Process (computing); Convergence (economics); Machine learning; Unsupervised learning; Reinforcement; Learning classifier system; Sequence learning; Psychology; Social psychology","score_opus":0.10468339204744868,"score_gpt":0.19147103244595107,"score_spread":0.08678764039850238,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3014454071","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006828913,0.0006822251,0.98691213,0.00027447942,0.00012071919,0.00005822103,0.000038523758,0.0005535597,0.0045312177],"genre_scores_gemma":[0.7770846,0.0010287978,0.20948605,0.0005678376,0.00019104639,0.0003995378,0.00017060767,0.00019851381,0.010872994],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990238,0.0003901314,0.000055639044,0.00018626006,0.00024299494,0.00010122912],"domain_scores_gemma":[0.9970632,0.0017181745,0.0002439316,0.00032855308,0.0005014105,0.00014463045],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018169103,0.0010408771,0.0013980523,0.00042626922,0.0004124671,0.0009790116,0.0020867088,0.0014095556,0.0042509562],"category_scores_gemma":[0.006899756,0.00045930434,0.0005324419,0.0005453159,0.0018173531,0.0015007907,0.0021624258,0.001836678,0.00092848344],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023064637,0.00018984397,0.0011751902,0.00041647922,0.00011733665,0.0002002066,0.00017602091,0.6760323,0.0036356878,0.105874486,0.0054295054,0.20652223],"study_design_scores_gemma":[0.000025584673,0.0000871108,0.000084746665,0.000014727183,0.000010365643,0.00003315526,0.0000052018436,0.9735622,0.0005980404,0.023557378,0.0020116614,0.000009832402],"about_ca_topic_score_codex":0.002758517,"about_ca_topic_score_gemma":0.0020874448,"teacher_disagreement_score":0.0042509562,"about_ca_system_score_codex":0.0007983019,"about_ca_system_score_gemma":0.0014788242,"threshold_uncertainty_score":0.014220834},"labels":[],"label_agreement":null},{"id":"W3015100386","doi":"10.48550/arxiv.2004.00600","title":"Work in Progress: Temporally Extended Auxiliary Tasks","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Robustness (evolution); Reinforcement learning; Task (project management); Autoencoder; Trajectory; Sensitivity (control systems); Artificial intelligence; Machine learning; Algorithm; Deep learning; Engineering","score_opus":0.08161985637333861,"score_gpt":0.2053707032589729,"score_spread":0.12375084688563427,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3015100386","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19905978,0.0015609477,0.7907121,0.0009680566,0.00039649254,0.00015342845,0.00016143527,0.0017242389,0.0052634915],"genre_scores_gemma":[0.85378116,0.0003909299,0.14265616,0.00034223098,0.000116361516,0.00011075332,0.00024330775,0.00015145374,0.0022076787],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992483,0.00027668435,0.000048217735,0.00017479362,0.0001743895,0.00007757248],"domain_scores_gemma":[0.993508,0.0036289126,0.000335045,0.0012993678,0.0009109296,0.00031782093],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029195633,0.00094902853,0.0009582179,0.00025463058,0.00031916497,0.0009564483,0.0014150288,0.0010233052,0.0034647172],"category_scores_gemma":[0.010544125,0.00026738312,0.0005188591,0.00032348276,0.00079846586,0.0023108062,0.0012415141,0.0023607137,0.00062864466],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010812988,0.0009150148,0.0036805957,0.00025597538,0.00014849345,0.00013822432,0.00024637228,0.67852616,0.0179341,0.013002328,0.0038354225,0.28023604],"study_design_scores_gemma":[0.00004542799,0.00021759962,0.00048257047,0.000016950935,0.000016053567,0.00002759652,0.000015358106,0.99055207,0.003263148,0.0040648957,0.0012857327,0.000012567168],"about_ca_topic_score_codex":0.0027978136,"about_ca_topic_score_gemma":0.002064684,"teacher_disagreement_score":0.0034647172,"about_ca_system_score_codex":0.00045989142,"about_ca_system_score_gemma":0.0011446592,"threshold_uncertainty_score":0.015440345},"labels":[],"label_agreement":null},{"id":"W3015230978","doi":"10.48550/arxiv.2004.03761","title":"Adaptive Transformers in RL","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Transformer; Computer science; Computation; Replicate; Reinforcement learning; Artificial intelligence; Computer engineering; Machine learning; Voltage; Electrical engineering; Engineering; Algorithm","score_opus":0.1102922021503649,"score_gpt":0.18820508747018033,"score_spread":0.07791288531981543,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3015230978","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012650768,0.00034871075,0.979784,0.00035143617,0.000077431774,0.000040451054,0.000090744565,0.001498509,0.0051578227],"genre_scores_gemma":[0.8552926,0.00055452157,0.13575958,0.00032351838,0.000089229776,0.00015822747,0.000182575,0.0004510336,0.0071888277],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991849,0.00029461927,0.000057059613,0.00021623263,0.00016332963,0.00008393366],"domain_scores_gemma":[0.99801993,0.0013211739,0.00011555964,0.0002669854,0.0001875262,0.00008884735],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001143246,0.0009228398,0.0007255414,0.00047341388,0.0003137576,0.0012564977,0.0010794413,0.0008217589,0.0065158876],"category_scores_gemma":[0.006698966,0.00044601105,0.0005709123,0.00038260766,0.0018223373,0.0022007234,0.001884173,0.0018916586,0.0012283337],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002865435,0.000056633042,0.00085880497,0.00024678238,0.00007082068,0.00017592074,0.00024502203,0.6021633,0.008726371,0.2604078,0.0036732913,0.12308873],"study_design_scores_gemma":[0.000033360826,0.000056594865,0.00008105503,0.000016514745,0.000013755044,0.00003915302,0.000018496552,0.84172493,0.0021941846,0.1539668,0.0018439031,0.000011320482],"about_ca_topic_score_codex":0.0028268625,"about_ca_topic_score_gemma":0.0029986654,"teacher_disagreement_score":0.0065158876,"about_ca_system_score_codex":0.0009250402,"about_ca_system_score_gemma":0.0010380277,"threshold_uncertainty_score":0.021797836},"labels":[],"label_agreement":null},{"id":"W3015239645","doi":"10.1007/s12065-020-00394-9","title":"Neuromodulated multiobjective evolutionary neurocontrollers without speciation","year":2020,"lang":"en","type":"article","venue":"Evolutionary Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Neuroevolution; Computer science; Evolutionary algorithm; Artificial intelligence; Neuromodulation; Genetic algorithm; Artificial neural network; Mathematical optimization; Mathematics; Ecology; Biology","score_opus":0.026577096686810284,"score_gpt":0.24533899506254947,"score_spread":0.21876189837573917,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3015239645","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17949554,0.00051646295,0.7897199,0.00038838424,0.00031750763,0.00010926146,0.000043234253,0.0007415586,0.028668208],"genre_scores_gemma":[0.9422893,0.00006649937,0.050638255,0.0001166069,0.000016134894,0.00008741094,0.000018268014,0.00003437963,0.006733203],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99989605,0.000022338325,0.0000057633224,0.000027039225,0.000028763952,0.000020128638],"domain_scores_gemma":[0.99976045,0.00008140195,0.00003517973,0.00003977416,0.000060973933,0.000022208113],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00037323017,0.00049103145,0.00048090718,0.0002412871,0.00033523282,0.000716174,0.0012637983,0.0010675727,0.0034918296],"category_scores_gemma":[0.0010942572,0.00027560553,0.0003340677,0.00016077868,0.00052380114,0.00057789776,0.0008217714,0.000789002,0.00040433017],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011407183,0.00010321317,0.00044425888,0.00007129935,0.000075815144,0.00014093515,0.000060823222,0.88842213,0.01689199,0.017760556,0.00082014623,0.07509477],"study_design_scores_gemma":[0.000017193093,0.000062801504,0.00009670826,0.0000074211575,0.000009887147,0.00003493804,0.0000066990438,0.9944839,0.0018154489,0.0029617134,0.0004973196,0.000005990701],"about_ca_topic_score_codex":0.00084054767,"about_ca_topic_score_gemma":0.0015423953,"teacher_disagreement_score":0.0034918296,"about_ca_system_score_codex":0.00044247802,"about_ca_system_score_gemma":0.00037256238,"threshold_uncertainty_score":0.011681318},"labels":[],"label_agreement":null},{"id":"W3015837232","doi":"10.1109/icra40945.2020.9196879","title":"Learning to Drive Off Road on Smooth Terrain in Unstructured Environments Using an On-Board Camera and Sparse Aerial Images","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"McGill University","funders":"","keywords":"Terrain; Computer science; Tree traversal; Artificial intelligence; Computer vision; Overhead (engineering); Aerial image; Image (mathematics); Remote sensing; Geology; Geography; Algorithm; Cartography","score_opus":0.027105756007752637,"score_gpt":0.27052294769151775,"score_spread":0.2434171916837651,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3015837232","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.146072,0.000106041996,0.8477256,0.00025814315,0.000041630214,0.00008282016,0.000114272305,0.0019477288,0.0036517666],"genre_scores_gemma":[0.9211002,0.00004326803,0.075846806,0.000074993455,0.000014384646,0.00006072941,0.00013775963,0.00007660432,0.0026451855],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99979764,0.000039019484,0.0000072222024,0.000071526476,0.000042912223,0.000041691394],"domain_scores_gemma":[0.99954045,0.00018581422,0.000072059236,0.00007420277,0.000062546525,0.00006490773],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003772511,0.00080411235,0.00058049243,0.0003027072,0.00030745653,0.0004627218,0.0012480482,0.0007330292,0.0016197436],"category_scores_gemma":[0.0012306272,0.00046001663,0.0004894886,0.00020472756,0.00062976877,0.0008093377,0.0008547088,0.00087440317,0.00039255095],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007884384,0.00009996134,0.0020641906,0.00003519738,0.000035302535,0.00007879829,0.00006315338,0.92376524,0.0048261043,0.001462773,0.0010234489,0.066466905],"study_design_scores_gemma":[0.0000069424627,0.00003055878,0.0001471518,0.0000020901566,0.0000031787033,0.000010599589,0.0000073898464,0.99820244,0.0007159941,0.0006907085,0.00018030671,0.0000027236408],"about_ca_topic_score_codex":0.010089649,"about_ca_topic_score_gemma":0.013819077,"teacher_disagreement_score":0.010089649,"about_ca_system_score_codex":0.0005611652,"about_ca_system_score_gemma":0.0010587828,"threshold_uncertainty_score":0.020061791},"labels":[],"label_agreement":null},{"id":"W3017390413","doi":"","title":"Active Roll-outs in MDP with Irreversible Dynamics","year":2019,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Dynamics (music); Chemistry; Computer science; Physics","score_opus":0.009771959918208143,"score_gpt":0.21452481604992762,"score_spread":0.20475285613171948,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3017390413","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09194224,0.00039293172,0.902527,0.0006510981,0.000032810938,0.00010124588,0.00011548449,0.0010665763,0.003170747],"genre_scores_gemma":[0.94217426,0.00014292166,0.05496481,0.00014583746,0.000020633102,0.0001591139,0.0000932566,0.00006952378,0.0022296377],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992304,0.0002703441,0.000038809943,0.00018564714,0.00014583956,0.00012898237],"domain_scores_gemma":[0.99689204,0.0023044653,0.00030798052,0.00016901128,0.00013507047,0.00019131627],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019211674,0.0013062193,0.001319364,0.00047344074,0.00077888934,0.0010149083,0.0011456263,0.0013908254,0.002068227],"category_scores_gemma":[0.0044601723,0.00069585553,0.0006385542,0.00040073082,0.0018198496,0.0013917609,0.0017735629,0.0019001911,0.00027011117],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000070354574,0.000024748648,0.00028352215,0.00002265422,0.000011382049,0.000054783886,0.000035218487,0.98486024,0.00040284308,0.0073854965,0.00022085637,0.006627966],"study_design_scores_gemma":[0.000011728899,0.000015906047,0.000028124496,0.000002219417,0.000002708414,0.0000056958625,0.0000038387207,0.9950559,0.00023179555,0.004529639,0.00010980911,0.0000026797131],"about_ca_topic_score_codex":0.0056806877,"about_ca_topic_score_gemma":0.003947583,"teacher_disagreement_score":0.0056806877,"about_ca_system_score_codex":0.0013362259,"about_ca_system_score_gemma":0.0013823766,"threshold_uncertainty_score":0.011295259},"labels":[],"label_agreement":null},{"id":"W3020277389","doi":"10.48550/arxiv.2004.12399","title":"Reinforcement Learning Generalization with Surprise Minimization","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Surprise; Generalization; Computer science; Reinforcement learning; Artificial intelligence; Robustness (evolution); Machine learning; Mathematics; Psychology","score_opus":0.06140952338521455,"score_gpt":0.1835702624956942,"score_spread":0.12216073911047964,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3020277389","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27679625,0.00047620235,0.7084391,0.0018282682,0.00016504784,0.00029374904,0.00022945608,0.0019342025,0.009837671],"genre_scores_gemma":[0.95285505,0.000081435406,0.044521995,0.00032823163,0.000035379242,0.0001842026,0.00014846546,0.00011519191,0.0017301938],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998285,0.00063973904,0.00009996871,0.00043653007,0.00033345952,0.000205328],"domain_scores_gemma":[0.993646,0.0035669818,0.0006619377,0.0011748605,0.0005688991,0.000381311],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040333476,0.0014197398,0.0014500716,0.00035872994,0.00048816347,0.0010557919,0.001880234,0.0017711982,0.0019367731],"category_scores_gemma":[0.015638849,0.00045382528,0.000692883,0.0002476247,0.0021368028,0.0019032195,0.0022862866,0.0029338303,0.00036781162],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023500428,0.00019834802,0.0018977071,0.0001279702,0.00010931674,0.0001242161,0.00008630644,0.9375467,0.0038903793,0.014945121,0.0019894105,0.0388496],"study_design_scores_gemma":[0.000027763175,0.00014393055,0.00025774815,0.000010267175,0.0000097462835,0.000023798613,0.000010914967,0.97907376,0.0011522985,0.018955847,0.00032483574,0.0000091410075],"about_ca_topic_score_codex":0.0031059815,"about_ca_topic_score_gemma":0.0027159539,"teacher_disagreement_score":0.0040333476,"about_ca_system_score_codex":0.001660608,"about_ca_system_score_gemma":0.0014849098,"threshold_uncertainty_score":0.021330595},"labels":[],"label_agreement":null},{"id":"W3023876387","doi":"10.48550/arxiv.2005.01138","title":"Off-Policy Adversarial Inverse Reinforcement Learning","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Adversarial system; Reinforcement learning; Computer science; Artificial intelligence; Imitation; Machine learning; Task (project management); Function (biology); Transfer of learning; Class (philosophy); Temporal difference learning; Psychology; Engineering","score_opus":0.0758925450107277,"score_gpt":0.2042774652760245,"score_spread":0.1283849202652968,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3023876387","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019309096,0.00019403936,0.97438216,0.00027418794,0.00006536628,0.00006334683,0.0000419758,0.00083072355,0.004839227],"genre_scores_gemma":[0.9140307,0.00012750432,0.07924981,0.00026212612,0.00003937879,0.00014899418,0.00011957505,0.000106904816,0.005915111],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994407,0.00017786259,0.00002280707,0.00013853156,0.00013516743,0.000084901534],"domain_scores_gemma":[0.998028,0.0013052394,0.00016469909,0.00019349248,0.00021268631,0.00009582501],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010186246,0.0009468672,0.0010228934,0.0003045709,0.00029366638,0.00061133044,0.0012414586,0.0010117409,0.0033065088],"category_scores_gemma":[0.0046709846,0.00029607423,0.00037031405,0.00023037424,0.0012638867,0.000750736,0.0011959403,0.0016271485,0.00062208524],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000099499906,0.000063928266,0.0005027637,0.000049552807,0.000020193553,0.00009549586,0.000035269808,0.94746494,0.0015542993,0.011977351,0.0014508547,0.0366859],"study_design_scores_gemma":[0.000005828747,0.000013482591,0.000027468075,0.0000026173539,0.0000015098395,0.000010080607,0.0000020296332,0.9969502,0.00030220574,0.0024661282,0.00021637171,0.0000020874966],"about_ca_topic_score_codex":0.0028104952,"about_ca_topic_score_gemma":0.0020283523,"teacher_disagreement_score":0.0033065088,"about_ca_system_score_codex":0.00067841576,"about_ca_system_score_gemma":0.0010162765,"threshold_uncertainty_score":0.01106137},"labels":[],"label_agreement":null},{"id":"W3024138969","doi":"10.48550/arxiv.2005.06223","title":"DREAM Architecture: a Developmental Approach to Open-Ended Learning in Robotics","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Animal Health Institute","funders":"","keywords":"Dream; Architecture; Robotics; Artificial intelligence; Computer science; Developmental robotics; Cognitive science; Engineering; Psychology; Robot; Art; Visual arts; Neuroscience","score_opus":0.10023880800136796,"score_gpt":0.20699681502897768,"score_spread":0.10675800702760972,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3024138969","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0070837582,0.00023026092,0.9822275,0.0006482954,0.000027835453,0.00008776745,0.00004880924,0.0005403848,0.009105385],"genre_scores_gemma":[0.20485233,0.000421831,0.7865682,0.00022115457,0.000019729006,0.00044327107,0.00013284096,0.00015757374,0.0071830293],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99891436,0.00045458265,0.000065142885,0.00029204204,0.00020851148,0.000065377695],"domain_scores_gemma":[0.99806696,0.0009291178,0.00015250621,0.0004097767,0.00025757754,0.00018397883],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020667561,0.00061338214,0.00041839466,0.00063949364,0.0006118463,0.0019889977,0.0034179094,0.0014061788,0.004326859],"category_scores_gemma":[0.006528277,0.00060049514,0.0008673576,0.00037274466,0.0048929015,0.0032013466,0.0034330853,0.0029062582,0.0008362653],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006824804,0.00009346728,0.0012137193,0.00030917814,0.000048485002,0.00018445907,0.0017184617,0.053281184,0.0044159703,0.8338411,0.0017391695,0.1030866],"study_design_scores_gemma":[0.000031859196,0.00013860216,0.0003962501,0.00010977568,0.000030158963,0.00021183785,0.00023132624,0.18955418,0.0063389526,0.7758583,0.027057163,0.000041592783],"about_ca_topic_score_codex":0.002005465,"about_ca_topic_score_gemma":0.0026694606,"teacher_disagreement_score":0.004326859,"about_ca_system_score_codex":0.0017389496,"about_ca_system_score_gemma":0.0016423443,"threshold_uncertainty_score":0.01447475},"labels":[],"label_agreement":null},{"id":"W3025133396","doi":"10.65109/uazl2918","title":"META-Learning State-based Eligibility Traces for More Sample-Efficient Policy Evaluation","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Machine learning; Robustness (evolution); Artificial intelligence; Temporal difference learning; Sample (material); Q-learning; Bellman equation; Mathematical optimization; Mathematics","score_opus":0.16838999763150939,"score_gpt":0.39732922141276855,"score_spread":0.22893922378125917,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3025133396","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016011138,0.00014042915,0.9818901,0.00019801325,0.00003385989,0.000064583764,0.00003263368,0.0007923952,0.00083689386],"genre_scores_gemma":[0.7607452,0.00012339988,0.23658414,0.0002472543,0.00005642911,0.00029885294,0.00014862654,0.00025375804,0.0015423463],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989736,0.00035794883,0.00007713618,0.00021745093,0.0002627182,0.00011114571],"domain_scores_gemma":[0.9923741,0.00538676,0.0005475877,0.00074566517,0.000650155,0.000295628],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034609227,0.0011914144,0.0018438409,0.00080536155,0.00038943917,0.0015456694,0.0023031498,0.0015968044,0.003335312],"category_scores_gemma":[0.019216398,0.00071829016,0.0006225888,0.00057106494,0.0012639132,0.002603257,0.0019021807,0.0031440805,0.00061368267],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023314675,0.0002585044,0.001342223,0.00012921199,0.000056844514,0.00006470696,0.00012503557,0.8550237,0.0033165542,0.018782,0.0013146062,0.119353525],"study_design_scores_gemma":[0.00001408505,0.000028408307,0.000048064292,0.000008861103,0.0000042845527,0.000008353875,0.000004042738,0.9942358,0.0007361074,0.0047281287,0.00017984009,0.000003892285],"about_ca_topic_score_codex":0.0027563535,"about_ca_topic_score_gemma":0.0028842567,"teacher_disagreement_score":0.0034609227,"about_ca_system_score_codex":0.0013084362,"about_ca_system_score_gemma":0.0027842156,"threshold_uncertainty_score":0.018303335},"labels":[],"label_agreement":null},{"id":"W3029466658","doi":"10.1109/tg.2020.2990865","title":"Winning Is Not Everything: Enhancing Game Development With Intelligent Agents","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Games","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Electronic Arts (Canada)","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Benchmark (surveying); Hyperparameter; Human–computer interaction; Machine learning","score_opus":0.043264560459594814,"score_gpt":0.25315020327216015,"score_spread":0.20988564281256533,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3029466658","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34995753,0.0007043677,0.6245953,0.0013741645,0.00009439177,0.00048289535,0.000059682665,0.0019946534,0.02073695],"genre_scores_gemma":[0.7635653,0.00027507133,0.23313521,0.00024983627,0.000016026212,0.00020023646,0.00007527063,0.00009044553,0.002392551],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99910396,0.0005088502,0.000041133913,0.000110908506,0.00014414081,0.00009099233],"domain_scores_gemma":[0.99788374,0.0012490549,0.00022112807,0.00021317294,0.00020022955,0.00023261571],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021313964,0.0008380489,0.00032042258,0.0004207164,0.000349967,0.0011836461,0.0014636682,0.00081544445,0.0016707925],"category_scores_gemma":[0.008161763,0.0003247837,0.00027314553,0.00020132525,0.0008793921,0.0017975946,0.0021913522,0.0011375628,0.00041263623],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055962073,0.0016964796,0.013721228,0.00036662223,0.00015993408,0.00035957003,0.0014299006,0.54291266,0.0194825,0.038486898,0.005059768,0.3757648],"study_design_scores_gemma":[0.00010359038,0.00050772075,0.0014171285,0.00006124091,0.000047748217,0.000105878295,0.00023006083,0.9550388,0.008897657,0.02358305,0.009979183,0.000027909171],"about_ca_topic_score_codex":0.0013884915,"about_ca_topic_score_gemma":0.0022811075,"teacher_disagreement_score":0.0021313964,"about_ca_system_score_codex":0.00055428623,"about_ca_system_score_gemma":0.0007049607,"threshold_uncertainty_score":0.011272073},"labels":[],"label_agreement":null},{"id":"W3034607397","doi":"","title":"An Optimistic Perspective on Offline Deep Reinforcement Learning","year":2020,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":95,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Google (Canada); University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Perspective (graphical); Artificial intelligence; Reinforcement; Machine learning; Psychology; Social psychology","score_opus":0.04466082375581626,"score_gpt":0.3204855486435333,"score_spread":0.27582472488771703,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3034607397","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011701743,0.018407717,0.72291636,0.13816041,0.0039702062,0.000031348594,0.00034469628,0.00032846635,0.10413905],"genre_scores_gemma":[0.8172453,0.015704723,0.09115303,0.013350904,0.011361647,0.00015305397,0.00022848143,0.00030694413,0.050495874],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99828774,0.0007985032,0.00005257446,0.00021207056,0.0005228274,0.00012631746],"domain_scores_gemma":[0.9909609,0.0066884495,0.0002884717,0.0008350075,0.0008856789,0.000341529],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035846038,0.0010833823,0.00091436185,0.0005724762,0.0008744919,0.003424035,0.0018503168,0.0031655636,0.0076192017],"category_scores_gemma":[0.01386805,0.00046354532,0.000390664,0.00061226595,0.003994334,0.0084235575,0.002499212,0.008672443,0.0011314309],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013162215,0.000039392537,0.00012987065,0.00010623961,0.000025740654,0.00005015662,0.000059512673,0.02138823,0.00026620092,0.9258012,0.011114414,0.04088736],"study_design_scores_gemma":[0.000019050512,0.00002473404,0.000052346884,0.000051904462,0.0000075126927,0.000030589454,0.00002137016,0.047322705,0.00030277582,0.9422253,0.00992963,0.000012175352],"about_ca_topic_score_codex":0.0015095127,"about_ca_topic_score_gemma":0.0010772039,"teacher_disagreement_score":0.0076192017,"about_ca_system_score_codex":0.0022376785,"about_ca_system_score_gemma":0.0012133087,"threshold_uncertainty_score":0.025488794},"labels":[],"label_agreement":null},{"id":"W3034724428","doi":"10.48550/arxiv.2003.00203","title":"Contextual Policy Transfer in Reinforcement Learning Domains via Deep Mixtures-of-Experts","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Reuse; Transfer of learning; Artificial intelligence; Robustness (evolution); Task (project management); Context (archaeology); Machine learning; Policy learning; Dynamics (music); Engineering","score_opus":0.053756362553908234,"score_gpt":0.20865636249188882,"score_spread":0.1548999999379806,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3034724428","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038698077,0.00041824303,0.9580233,0.00035463547,0.00004463429,0.000064425694,0.000048469192,0.0008407869,0.0015074235],"genre_scores_gemma":[0.9037486,0.00019781837,0.09321253,0.00022503316,0.000044881694,0.00014837444,0.00010523512,0.00013134498,0.0021861694],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99868053,0.000625058,0.00005812609,0.00028644872,0.0001968433,0.00015300565],"domain_scores_gemma":[0.9963368,0.0026278514,0.00025757035,0.00029069485,0.0002763793,0.00021060447],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00336829,0.001347436,0.0016524476,0.00054643396,0.00045936,0.0010549352,0.002219943,0.0017879647,0.0020606664],"category_scores_gemma":[0.011203269,0.0009033659,0.00076211523,0.00051513064,0.0018452308,0.0022876891,0.0025634447,0.0031981433,0.00043724247],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012883187,0.0000750681,0.0005969953,0.000046469744,0.000040747505,0.00004620099,0.000103691134,0.96176064,0.000687609,0.009260635,0.0005342958,0.026718875],"study_design_scores_gemma":[0.000011241231,0.000021870705,0.000041568863,0.000004690473,0.0000039872757,0.0000050590797,0.0000046952664,0.9922856,0.0002744982,0.007197294,0.00014525143,0.0000042696165],"about_ca_topic_score_codex":0.006858031,"about_ca_topic_score_gemma":0.005853498,"teacher_disagreement_score":0.006858031,"about_ca_system_score_codex":0.0016203298,"about_ca_system_score_gemma":0.0014751027,"threshold_uncertainty_score":0.017813444},"labels":[],"label_agreement":null},{"id":"W3034739078","doi":"10.1109/isorc49007.2020.00033","title":"Context-based learning for autonomous vehicles","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Computer science; Context (archaeology); Artificial intelligence; Control (management); Deep learning; Machine learning; Human–computer interaction","score_opus":0.03608482396398357,"score_gpt":0.25076122593306166,"score_spread":0.21467640196907808,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3034739078","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04322546,0.0019100115,0.94716656,0.0007448336,0.00020826863,0.000053682812,0.000093033406,0.00069226156,0.005905859],"genre_scores_gemma":[0.9338431,0.00056350866,0.062074307,0.00019881592,0.00005774677,0.00006170983,0.000114340684,0.000042002655,0.0030443575],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99983215,0.000036183483,0.000007635897,0.000058503898,0.000036779726,0.000028627093],"domain_scores_gemma":[0.9998054,0.000083447674,0.000024697612,0.000016665199,0.00004725232,0.000022631688],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003396557,0.00038888823,0.00040821586,0.00021104136,0.00029965604,0.00044192348,0.00078795163,0.00053266296,0.0017813465],"category_scores_gemma":[0.001150541,0.00022317408,0.00028012364,0.00021452316,0.00048466885,0.0007500323,0.0009460769,0.0010305214,0.00024394422],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000628698,0.000052639163,0.0008762429,0.00008910727,0.000029860465,0.0000911383,0.00007962122,0.87417805,0.003916723,0.023545811,0.0020971927,0.09498082],"study_design_scores_gemma":[0.0000050408707,0.000024150326,0.0001185484,0.0000054824272,0.0000037448854,0.000010074778,0.000010487257,0.9873582,0.00036242956,0.010876311,0.0012212789,0.000004240323],"about_ca_topic_score_codex":0.009456787,"about_ca_topic_score_gemma":0.010086701,"teacher_disagreement_score":0.009456787,"about_ca_system_score_codex":0.00068869005,"about_ca_system_score_gemma":0.0008145522,"threshold_uncertainty_score":0.018803477},"labels":[],"label_agreement":null},{"id":"W3034833014","doi":"","title":"OPtions as REsponses: Grounding behavioural hierarchies in multi-agent reinforcement learning","year":2020,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Ground; Reinforcement; Artificial intelligence; Cognitive psychology; Human–computer interaction; Cognitive science; Psychology; Engineering; Social psychology; Electrical engineering","score_opus":0.12890161320820673,"score_gpt":0.34831895195492124,"score_spread":0.2194173387467145,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3034833014","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14239797,0.0001437292,0.84984916,0.0007890675,0.000060914863,0.000084182124,0.000098004886,0.0005121082,0.0060649207],"genre_scores_gemma":[0.94391453,0.000045963418,0.054494325,0.000101266545,0.0000109462935,0.00007879031,0.00005388874,0.000045046945,0.0012552586],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99897885,0.0005755264,0.000044759992,0.00016832085,0.000113357484,0.000119210614],"domain_scores_gemma":[0.9942457,0.0038769776,0.00047240622,0.0005877593,0.0004025082,0.00041466384],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023558596,0.0005889152,0.00079772243,0.0005255658,0.0005544263,0.0012519074,0.0017835048,0.0012953947,0.0048223543],"category_scores_gemma":[0.012079798,0.0006077595,0.0005185204,0.00037897303,0.0021540136,0.002576538,0.0023353635,0.0021158282,0.00028842848],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025146958,0.00014862166,0.002437349,0.000112391666,0.00006847967,0.00011568786,0.0005798488,0.8209447,0.0028049995,0.11653964,0.00093374477,0.055063024],"study_design_scores_gemma":[0.000024873681,0.000037070302,0.00017847568,0.000010560432,0.00000810479,0.000006519702,0.000041401356,0.92714983,0.0002646463,0.07206038,0.00020952443,0.00000854415],"about_ca_topic_score_codex":0.004308442,"about_ca_topic_score_gemma":0.0052290154,"teacher_disagreement_score":0.0048223543,"about_ca_system_score_codex":0.0008961641,"about_ca_system_score_gemma":0.0008729218,"threshold_uncertainty_score":0.016132355},"labels":[],"label_agreement":null},{"id":"W3035417664","doi":"10.48550/arxiv.2006.07262","title":"A Brief Look at Generalization in Visual Meta-Reinforcement Learning","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Overfitting; Generalization; Computer science; Artificial intelligence; Meta learning (computer science); Scalability; Machine learning; Realization (probability); Artificial neural network; Task (project management); Mathematics","score_opus":0.1025356039800561,"score_gpt":0.2165601146297081,"score_spread":0.114024510649652,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3035417664","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02544777,0.005929979,0.9575396,0.0019394826,0.00022915061,0.000082235085,0.00007058325,0.00081314,0.007948159],"genre_scores_gemma":[0.81920666,0.0042882743,0.16886246,0.0011355489,0.00042013687,0.0002742388,0.00013797286,0.00033713196,0.005337671],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988626,0.0004385014,0.000068234185,0.00029731332,0.0002258398,0.00010747773],"domain_scores_gemma":[0.99679416,0.0019354838,0.00031103578,0.00055649376,0.00027647978,0.00012620992],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002981539,0.0010504188,0.0011167709,0.0005194378,0.00041562848,0.0014803206,0.0019759096,0.0017526521,0.0028482731],"category_scores_gemma":[0.011217851,0.0005698932,0.0013610545,0.0006693693,0.0020772112,0.003079072,0.0021021343,0.0041176323,0.00043338127],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010421603,0.00010165974,0.0020513607,0.00032769257,0.00017473602,0.00016049779,0.00029639556,0.7945491,0.0047862856,0.09350996,0.0029237182,0.101014346],"study_design_scores_gemma":[0.000012309448,0.0001937299,0.00064533536,0.00010134736,0.000020857244,0.00009673187,0.000037156744,0.92042816,0.0014657036,0.07363228,0.0033392026,0.000027246435],"about_ca_topic_score_codex":0.003263012,"about_ca_topic_score_gemma":0.0018242917,"teacher_disagreement_score":0.003263012,"about_ca_system_score_codex":0.0013635111,"about_ca_system_score_gemma":0.00072772894,"threshold_uncertainty_score":0.01576811},"labels":[],"label_agreement":null},{"id":"W3035954878","doi":"10.48550/arxiv.2006.13169","title":"Experience Replay with Likelihood-free Importance Weights","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Estimator; Temporal difference learning; Suite; Key (lock); Baseline (sea); Bellman equation; Function (biology); Sample (material); Artificial intelligence; Prioritization; Machine learning; Mathematical optimization; Statistics; Mathematics","score_opus":0.05575956341198271,"score_gpt":0.18530986196650417,"score_spread":0.12955029855452146,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3035954878","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.056337196,0.00022026108,0.94047827,0.00026175517,0.00008032285,0.000080952515,0.000043184144,0.00082232157,0.0016757754],"genre_scores_gemma":[0.8809198,0.00007971282,0.11581787,0.00011691131,0.00003291276,0.00012297492,0.00008033373,0.00008594806,0.0027435725],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993297,0.00019881585,0.000042980828,0.0001642628,0.00017855695,0.00008568219],"domain_scores_gemma":[0.9973793,0.0013259124,0.00031827894,0.00037047127,0.00040027444,0.00020571926],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018501114,0.00094874506,0.0009754075,0.00040516822,0.00033771692,0.0009116239,0.0018699104,0.0010206257,0.0025780604],"category_scores_gemma":[0.008772,0.00054470106,0.0004121794,0.00035382563,0.0010661853,0.0017560316,0.0018918196,0.0022854088,0.00047119817],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000546072,0.00023375991,0.0022878465,0.00015517272,0.00007806326,0.00015007032,0.00016291518,0.82393086,0.0065644113,0.015756005,0.0018447809,0.14829001],"study_design_scores_gemma":[0.000022962102,0.000057987316,0.0001364824,0.0000062629765,0.000006330436,0.000015718491,0.000006982532,0.99431807,0.0012406167,0.003925484,0.00025747047,0.000005670574],"about_ca_topic_score_codex":0.0031978958,"about_ca_topic_score_gemma":0.003901338,"teacher_disagreement_score":0.0031978958,"about_ca_system_score_codex":0.0007850429,"about_ca_system_score_gemma":0.001322819,"threshold_uncertainty_score":0.00978446},"labels":[],"label_agreement":null},{"id":"W3035992805","doi":"10.1109/icra48506.2021.9560922","title":"LEAF: Latent Exploration Along the Frontier","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Artificial intelligence; State (computer science); Reachability; Oracle; Robot; Set (abstract data type); Key (lock); Machine learning; Frontier; Task (project management); State space; Theoretical computer science; Algorithm; Mathematics; Geography","score_opus":0.04681003592202214,"score_gpt":0.25875255073531644,"score_spread":0.2119425148132943,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3035992805","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025727661,0.0002326523,0.96809644,0.00029391164,0.000029539182,0.00006317799,0.00019530584,0.0028203076,0.0025410268],"genre_scores_gemma":[0.7662809,0.000204517,0.2262038,0.00024148832,0.000032004504,0.0002295568,0.0007239025,0.00039785053,0.00568605],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999681,0.00007657418,0.000012793204,0.0001182822,0.00006632746,0.00004505304],"domain_scores_gemma":[0.99912924,0.00040254995,0.000115929295,0.00019024927,0.00008377464,0.000078384255],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00065276236,0.00060819444,0.0006856981,0.00034136663,0.0003531257,0.0007221709,0.0015338259,0.0009811929,0.0042292373],"category_scores_gemma":[0.0028477274,0.00037578217,0.0005548555,0.0002633066,0.0010518894,0.0020268045,0.0019103041,0.0017079674,0.0008890794],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003412829,0.0001747095,0.0024362132,0.0002177102,0.000064969085,0.00014782122,0.00022189706,0.74078685,0.0077920468,0.03808793,0.0070201186,0.20270848],"study_design_scores_gemma":[0.00001604304,0.00004496019,0.00012908125,0.000011566692,0.000004232395,0.000023269096,0.00001026874,0.9845464,0.0012239718,0.013136457,0.0008481457,0.000005621211],"about_ca_topic_score_codex":0.0022682024,"about_ca_topic_score_gemma":0.0035068095,"teacher_disagreement_score":0.0042292373,"about_ca_system_score_codex":0.00077471294,"about_ca_system_score_gemma":0.0012096893,"threshold_uncertainty_score":0.014148235},"labels":[],"label_agreement":null},{"id":"W3037098147","doi":"","title":"Efficient Planning under Partial Observability with Unnormalized Q Functions and Spectral Learning","year":2019,"lang":"en","type":"article","venue":"International Conference on Artificial Intelligence and Statistics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; McGill University","funders":"","keywords":"Observability; Sample complexity; Reinforcement learning; Computer science; Observable; Artificial intelligence; Sample (material); Plan (archaeology); Mathematical optimization; Algorithm; Machine learning; Mathematics","score_opus":0.09226570585973484,"score_gpt":0.3228543093033689,"score_spread":0.23058860344363405,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3037098147","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00740631,0.00004949765,0.99148995,0.00009626046,0.00000829021,0.000018881488,0.000011504964,0.00015855342,0.00076084887],"genre_scores_gemma":[0.6436853,0.00015389925,0.35403052,0.00011791473,0.000042713953,0.00018829606,0.00009593479,0.00011521312,0.0015702291],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987765,0.00056560093,0.000058695998,0.00021337382,0.00027306678,0.000112763984],"domain_scores_gemma":[0.99506617,0.003540054,0.00046105424,0.00045217117,0.00031283038,0.00016770813],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025108939,0.00080220285,0.0009370579,0.00064058707,0.00046504816,0.0010358656,0.001326261,0.0011518185,0.0019016204],"category_scores_gemma":[0.009926554,0.00049357227,0.0005258848,0.0006821807,0.0024188568,0.0029208364,0.0018931194,0.0013551441,0.00027485794],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000925702,0.00007247399,0.00039358865,0.00008474328,0.000026381018,0.000044679364,0.000094278934,0.8102442,0.0012171755,0.12693937,0.0006857804,0.060104784],"study_design_scores_gemma":[0.000008626215,0.00001934479,0.00004241148,0.0000056049853,0.000002109092,0.000008928393,0.0000065480313,0.9540047,0.00038715487,0.04533071,0.00017992806,0.0000039671527],"about_ca_topic_score_codex":0.0032193887,"about_ca_topic_score_gemma":0.0024723234,"teacher_disagreement_score":0.0032193887,"about_ca_system_score_codex":0.0012373698,"about_ca_system_score_gemma":0.002518277,"threshold_uncertainty_score":0.013279021},"labels":[],"label_agreement":null},{"id":"W3037179286","doi":"10.65109/rtpw2660","title":"Improving Performance in Reinforcement Learning by Breaking Generalization in Neural Networks","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Generalization; Scalability; Artificial neural network; Machine learning","score_opus":0.01277579907890778,"score_gpt":0.21482141115209308,"score_spread":0.2020456120731853,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3037179286","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1361522,0.00077411265,0.85563743,0.0006259442,0.00011020222,0.000117867894,0.000053473534,0.0024180193,0.004110711],"genre_scores_gemma":[0.93405336,0.00016307182,0.06412128,0.00027510696,0.00004542506,0.0001254652,0.00005802473,0.00013298495,0.0010253369],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982462,0.000504062,0.00013534377,0.00045282394,0.00041550174,0.00024611133],"domain_scores_gemma":[0.9927664,0.0043665073,0.0007095444,0.0013380633,0.000597645,0.00022184916],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004137988,0.0016458068,0.0013482383,0.00041675285,0.0005727106,0.0009118906,0.0017683291,0.001358973,0.0017926004],"category_scores_gemma":[0.019209448,0.00067091297,0.00066690345,0.00031352259,0.0020077187,0.0020547563,0.0020047878,0.003505345,0.00041453735],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021879816,0.00021143981,0.0021985257,0.000117177,0.00010094393,0.000095279494,0.000114275426,0.8874491,0.008168798,0.0073190415,0.0009903159,0.0930163],"study_design_scores_gemma":[0.000019833667,0.00012712537,0.0002687812,0.000010858692,0.000011226486,0.00001908912,0.000006847211,0.99249166,0.0019033143,0.0048895795,0.00024411784,0.0000075463267],"about_ca_topic_score_codex":0.0049988395,"about_ca_topic_score_gemma":0.0033006289,"teacher_disagreement_score":0.0049988395,"about_ca_system_score_codex":0.0014522148,"about_ca_system_score_gemma":0.0013300083,"threshold_uncertainty_score":0.021884084},"labels":[],"label_agreement":null},{"id":"W3037476194","doi":"10.1609/icaps.v30i1.6750","title":"Symbolic Plans as High-Level Instructions for Reinforcement Learning","year":2020,"lang":"en","type":"article","venue":"Proceedings of the International Conference on Automated Planning and Scheduling","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":63,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Agencia Nacional de Investigación y Desarrollo","keywords":"Reinforcement learning; Computer science; Maximization; Task (project management); State (computer science); Action (physics); Artificial intelligence; Machine learning; Mathematical optimization; Programming language; Mathematics","score_opus":0.06793028415246624,"score_gpt":0.2946574244530109,"score_spread":0.22672714030054467,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3037476194","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0052887355,0.000078589524,0.9906295,0.00016310209,0.000023125052,0.00007829276,0.00014442192,0.0011600162,0.0024341128],"genre_scores_gemma":[0.32125375,0.0002858556,0.67351407,0.00013645689,0.000034655215,0.00073300884,0.000551039,0.0003549987,0.0031362632],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992112,0.00031612845,0.00005373417,0.00010008198,0.00025568777,0.00006314241],"domain_scores_gemma":[0.9979942,0.0013510664,0.0001562772,0.00024699362,0.00017275491,0.000078727826],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009220642,0.0009569907,0.0005375133,0.00060971855,0.00040790992,0.0014802931,0.0014559459,0.001081959,0.0067497194],"category_scores_gemma":[0.005684379,0.0005982025,0.00067706226,0.000517505,0.0017944752,0.0016403134,0.0013229751,0.0021238574,0.001077303],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009262133,0.000074097064,0.0004924997,0.00017636367,0.0000248787,0.00014599382,0.00025756037,0.75723565,0.0035798112,0.17657292,0.0022208618,0.059126742],"study_design_scores_gemma":[0.000022698267,0.000021427357,0.00003591673,0.000025573141,0.0000067927976,0.000012233066,0.000018364322,0.91994303,0.001454306,0.075977094,0.00247506,0.0000075468433],"about_ca_topic_score_codex":0.0034269711,"about_ca_topic_score_gemma":0.0065199225,"teacher_disagreement_score":0.0067497194,"about_ca_system_score_codex":0.0013063207,"about_ca_system_score_gemma":0.0019198176,"threshold_uncertainty_score":0.022580087},"labels":[],"label_agreement":null},{"id":"W3037571487","doi":"","title":"Value Preserving State-Action Abstractions","year":2020,"lang":"en","type":"article","venue":"International Conference on Artificial Intelligence and Statistics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Value (mathematics); State (computer science); Action (physics); Programming language; Machine learning","score_opus":0.23789144969792853,"score_gpt":0.3741205699300081,"score_spread":0.13622912023207956,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3037571487","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006751448,0.00008771121,0.98788935,0.00016660831,0.00005748206,0.000046610672,0.00014347106,0.00079042435,0.0040669465],"genre_scores_gemma":[0.55993116,0.00036854736,0.42448342,0.00030162593,0.00007328342,0.00023366176,0.00055621006,0.000378942,0.013673147],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988556,0.00027743177,0.00009480288,0.00022814867,0.00039359022,0.00015052779],"domain_scores_gemma":[0.998218,0.0007403555,0.00011940496,0.00060357,0.00018172423,0.00013703392],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017892383,0.0008924238,0.00075511914,0.0006508761,0.00059531315,0.0019415478,0.0016619508,0.0010191668,0.0067228526],"category_scores_gemma":[0.004625426,0.000615711,0.0015661347,0.0006931228,0.0015101942,0.0031873719,0.004262074,0.0033693577,0.0010432567],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021931536,0.00012245706,0.0005583838,0.00017838989,0.00008169668,0.00024249573,0.00034662627,0.16097806,0.0058069746,0.7132933,0.002668574,0.11550374],"study_design_scores_gemma":[0.00003139029,0.00006846241,0.00012066489,0.000030718104,0.00004248289,0.00006115897,0.000045756835,0.41120642,0.004880922,0.5767396,0.0067554736,0.000016878845],"about_ca_topic_score_codex":0.001380471,"about_ca_topic_score_gemma":0.0020995461,"teacher_disagreement_score":0.0067228526,"about_ca_system_score_codex":0.0008753447,"about_ca_system_score_gemma":0.0013267279,"threshold_uncertainty_score":0.022490144},"labels":[],"label_agreement":null},{"id":"W3037719421","doi":"10.65109/gjmw6851","title":"Neural Replicator Dynamics: Multiagent Learning via Hedging Policy Gradients","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Softmax function; Reinforcement learning; Computer science; Nash equilibrium; Mathematical optimization; Replicator equation; Convergence (economics); Gradient descent; Best response; Regret; Margin (machine learning); Artificial intelligence; Artificial neural network; Applied mathematics; Mathematical economics; Mathematics; Machine learning; Economics","score_opus":0.017980739523340793,"score_gpt":0.2518673730577505,"score_spread":0.2338866335344097,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3037719421","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02724915,0.000317453,0.96869063,0.00036269208,0.000056517438,0.00006435017,0.00003270885,0.0004762481,0.0027503325],"genre_scores_gemma":[0.83944327,0.00022972756,0.15573221,0.0002557369,0.000037790218,0.00020270063,0.000071349285,0.000097619035,0.0039295293],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99958915,0.00017221348,0.000020472604,0.00009418538,0.000084220155,0.000039740506],"domain_scores_gemma":[0.9984351,0.0010586237,0.00015625083,0.00013542309,0.00013162446,0.00008298688],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001769704,0.00083955313,0.0010037262,0.000363666,0.000362438,0.0009510462,0.0016738359,0.0012058009,0.0023623686],"category_scores_gemma":[0.0054531293,0.00057473383,0.00041883203,0.0003236805,0.001103723,0.0015150323,0.0015482833,0.002043881,0.0003920272],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000060814342,0.00005208438,0.0008199678,0.00004652288,0.000052853953,0.000068537636,0.000084031344,0.9206044,0.00090771256,0.028982997,0.0010504611,0.047269657],"study_design_scores_gemma":[0.0000064644805,0.000010416148,0.000022816492,0.0000032173878,0.000002320914,0.0000061348933,0.0000025788931,0.994585,0.00015036885,0.0050546406,0.00015400817,0.0000021457804],"about_ca_topic_score_codex":0.0032150124,"about_ca_topic_score_gemma":0.0028696912,"teacher_disagreement_score":0.0032150124,"about_ca_system_score_codex":0.0009384217,"about_ca_system_score_gemma":0.001028395,"threshold_uncertainty_score":0.0093592405},"labels":[],"label_agreement":null},{"id":"W3037828233","doi":"10.48550/arxiv.1904.11439","title":"META-Learning State-based Eligibility Traces for More Sample-Efficient\\n Policy Evaluation","year":2019,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Robustness (evolution); Machine learning; Temporal difference learning; Artificial intelligence; Q-learning; Sample (material); TRACE (psycholinguistics)","score_opus":0.2079463326522307,"score_gpt":0.28507339274908255,"score_spread":0.07712706009685186,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3037828233","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025447762,0.00024033438,0.9704714,0.00037381312,0.00007159505,0.00009414996,0.00010966988,0.0018094868,0.0013818714],"genre_scores_gemma":[0.700618,0.00019947824,0.29436088,0.0004473236,0.00010305499,0.00040490457,0.0005419917,0.000543242,0.0027810365],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99894804,0.0003250362,0.000078513294,0.00027605752,0.00022637645,0.00014598607],"domain_scores_gemma":[0.993468,0.0045831082,0.000392959,0.0006311656,0.00062217104,0.00030270882],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031790638,0.0015230639,0.0022741982,0.0011598493,0.0005292966,0.0018782163,0.0032090035,0.0020654672,0.0057882126],"category_scores_gemma":[0.01783272,0.00090750895,0.00082754396,0.0008020511,0.0011892131,0.0033634861,0.0021502727,0.0036208956,0.0010227194],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032684393,0.0003130403,0.002181379,0.00016156495,0.00007700306,0.00008233662,0.00012992501,0.81407195,0.0026591162,0.01493612,0.002590417,0.16247034],"study_design_scores_gemma":[0.000011736251,0.000021718943,0.00005475373,0.000009698631,0.000004832868,0.000006154668,0.0000052654113,0.9953312,0.0005918614,0.0037667146,0.00019166945,0.000004360677],"about_ca_topic_score_codex":0.005145632,"about_ca_topic_score_gemma":0.006960625,"teacher_disagreement_score":0.0057882126,"about_ca_system_score_codex":0.0016681467,"about_ca_system_score_gemma":0.003383079,"threshold_uncertainty_score":0.019363463},"labels":[],"label_agreement":null},{"id":"W3037941652","doi":"10.65109/hmve9255","title":"Gifting in Multi-Agent Reinforcement Learning","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Function (biology); Reinforcement; Appropriation; Space (punctuation); Resource (disambiguation); Action (physics); Mechanism (biology); Property (philosophy); Error-driven learning; Artificial intelligence; Psychology; Social psychology","score_opus":0.06128900177664424,"score_gpt":0.27539004644848986,"score_spread":0.21410104467184563,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3037941652","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.104110375,0.0011222071,0.8839084,0.0011291313,0.000111202506,0.000069084876,0.00004597487,0.00028822914,0.0092154825],"genre_scores_gemma":[0.9631036,0.00033188285,0.034464065,0.00009224232,0.00004192603,0.00006997804,0.000016774355,0.00002246284,0.001857184],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986694,0.00078729534,0.000055994355,0.00017785147,0.00020442369,0.00010513201],"domain_scores_gemma":[0.99369097,0.004264402,0.0007545998,0.00043893646,0.00039593858,0.00045529136],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003038504,0.0006178961,0.0009924202,0.00043958705,0.00066256477,0.00097538426,0.0011230262,0.0012741534,0.0018201952],"category_scores_gemma":[0.014046143,0.00029127728,0.0004830302,0.00041537956,0.0025846618,0.0019871215,0.0014118495,0.0015204997,0.00017934148],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013042436,0.000075467855,0.0020023996,0.00014720095,0.000070941205,0.0001850209,0.00020294287,0.5913201,0.0016285178,0.37777987,0.00066942105,0.02578773],"study_design_scores_gemma":[0.00004426126,0.00007333881,0.000253366,0.000016223117,0.000012260318,0.000035609246,0.000021249123,0.80948687,0.00036813313,0.18864834,0.0010265522,0.000013844987],"about_ca_topic_score_codex":0.0015554529,"about_ca_topic_score_gemma":0.00095015677,"teacher_disagreement_score":0.003038504,"about_ca_system_score_codex":0.0014243732,"about_ca_system_score_gemma":0.00074092567,"threshold_uncertainty_score":0.016069353},"labels":[],"label_agreement":null},{"id":"W3037991458","doi":"","title":"What can I do here? A Theory of Affordances in Reinforcement Learning","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Affordance; Reinforcement learning; Computer science; Markov decision process; Embodied cognition; Context (archaeology); Dual (grammatical number); Function (biology); Artificial intelligence; Human–computer interaction; Cognitive science; Markov process; Mathematics; Psychology","score_opus":0.03236714438760317,"score_gpt":0.26857640851696307,"score_spread":0.23620926412935989,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3037991458","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01510011,0.0032007731,0.9367851,0.018028993,0.0003470934,0.000067470806,0.00020397748,0.00023901758,0.026027402],"genre_scores_gemma":[0.79048795,0.0034579535,0.19192158,0.0026750593,0.00056125096,0.00034103624,0.00023234438,0.00012436308,0.010198477],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9986241,0.00068915955,0.000057452122,0.0003087946,0.00021863975,0.00010176831],"domain_scores_gemma":[0.99596393,0.00304125,0.00029423778,0.0002196274,0.00026395157,0.00021698071],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024134365,0.00072872825,0.0007602387,0.00093014416,0.00095347816,0.0022998936,0.0013036893,0.00226651,0.0067116376],"category_scores_gemma":[0.016468987,0.00040193522,0.0008966723,0.00092433364,0.004157223,0.008183869,0.001479406,0.002562891,0.0009638747],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006760593,0.00005403169,0.0016185673,0.00014242365,0.000047221307,0.00014715448,0.00050883676,0.024509199,0.00032376332,0.9059971,0.004953938,0.061630104],"study_design_scores_gemma":[0.00001574519,0.000028708342,0.00022411461,0.000043341814,0.000009451526,0.00005436836,0.000080561695,0.05589935,0.00011222162,0.9389758,0.0045404797,0.00001592932],"about_ca_topic_score_codex":0.0031134496,"about_ca_topic_score_gemma":0.0021907089,"teacher_disagreement_score":0.0067116376,"about_ca_system_score_codex":0.001447772,"about_ca_system_score_gemma":0.0007589853,"threshold_uncertainty_score":0.022452652},"labels":[],"label_agreement":null},{"id":"W3037998275","doi":"10.65109/fyqq1225","title":"Inducing Cooperation through Reward Reshaping based on Peer Evaluations in Deep Multi-Agent Reinforcement Learning","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Selfishness; Reinforcement learning; Computer science; Function (biology); Task (project management); Action (physics); Artificial intelligence; Dual (grammatical number); Action selection; Value (mathematics); Machine learning; Psychology; Social psychology; Engineering","score_opus":0.1017844746347217,"score_gpt":0.33307506315892255,"score_spread":0.23129058852420084,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3037998275","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09008187,0.00020836818,0.9061229,0.00045826533,0.000053279127,0.00009611156,0.00001937289,0.00034628535,0.0026135626],"genre_scores_gemma":[0.9428162,0.000053463264,0.055218562,0.00014874336,0.000016242022,0.000111496534,0.00002272325,0.000033543038,0.0015788754],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99897885,0.0004505037,0.000046248082,0.00018401175,0.00019104638,0.00014948071],"domain_scores_gemma":[0.9973869,0.0013409138,0.00039943474,0.00020364547,0.00039174483,0.00027738628],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026423703,0.00091938203,0.0010887998,0.0003471437,0.00045189372,0.0006988895,0.0016696459,0.0011164837,0.0011577201],"category_scores_gemma":[0.0066062273,0.00039903485,0.00032047665,0.00022436118,0.0013006738,0.0011233023,0.0018142237,0.0014026967,0.00020318814],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014817949,0.00015583583,0.0020677615,0.000060033846,0.000062591644,0.00010594982,0.00017733578,0.92739254,0.0034933563,0.013460032,0.00090435427,0.051972006],"study_design_scores_gemma":[0.000014456109,0.000039459213,0.00008029201,0.0000039101446,0.0000051678535,0.00000977421,0.000007372588,0.9953803,0.00043004562,0.0038632916,0.00016169748,0.0000043062896],"about_ca_topic_score_codex":0.0024939184,"about_ca_topic_score_gemma":0.0029471712,"teacher_disagreement_score":0.0026423703,"about_ca_system_score_codex":0.0010672286,"about_ca_system_score_gemma":0.0012275279,"threshold_uncertainty_score":0.013974309},"labels":[],"label_agreement":null},{"id":"W3038104799","doi":"10.1109/lcsys.2020.3005886","title":"To Share or Not to Share? Performance Guarantees and the Asymmetric Nature of Cross-Robot Experience Transfer","year":2020,"lang":"en","type":"preprint","venue":"IEEE Control Systems Letters","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dynamic Systems Analysis (Canada); Vector Institute; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Ontario Research Foundation","keywords":"Robot; Computer science; Robotics; Transfer of learning; Artificial intelligence; Tracking (education); Bayesian probability; Transfer (computing); Inverse; Mathematics","score_opus":0.025791384409374694,"score_gpt":0.28392171397009963,"score_spread":0.25813032956072496,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3038104799","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05213143,0.0003341824,0.9358789,0.0011955986,0.00007350519,0.000056930454,0.000068351736,0.0004216922,0.009839427],"genre_scores_gemma":[0.9576985,0.00019497864,0.037199892,0.00034036348,0.00007490225,0.00013549626,0.000085080734,0.00016452672,0.0041062078],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9950794,0.001413959,0.00024916313,0.0010488618,0.0014054838,0.00080318924],"domain_scores_gemma":[0.9754346,0.011891098,0.0032566462,0.0060868594,0.002211092,0.0011196964],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006914138,0.0013214181,0.0015064935,0.00043239645,0.0010304006,0.0021281312,0.0024284835,0.0025913862,0.0063735913],"category_scores_gemma":[0.045044504,0.0005993067,0.00066771306,0.00047343105,0.0032911801,0.0071277334,0.0057923435,0.0038696446,0.0015221491],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011399826,0.00037973627,0.0036581608,0.0004711921,0.00013906571,0.00062367885,0.0007658739,0.5067535,0.019541139,0.28326154,0.003469095,0.17979695],"study_design_scores_gemma":[0.00006104391,0.00044021127,0.0013027128,0.00007336091,0.000026659465,0.00035734146,0.00017442477,0.7377135,0.009426068,0.24820217,0.0021728235,0.000049669827],"about_ca_topic_score_codex":0.00083305646,"about_ca_topic_score_gemma":0.00043997943,"teacher_disagreement_score":0.006914138,"about_ca_system_score_codex":0.0011439245,"about_ca_system_score_gemma":0.0013803495,"threshold_uncertainty_score":0.0365659},"labels":[],"label_agreement":null},{"id":"W3038500004","doi":"10.1007/s12652-021-03489-y","title":"A conceptual framework for externally-influenced agents: an assisted reinforcement learning review","year":2021,"lang":"en","type":"article","venue":"Journal of Ambient Intelligence and Humanized Computing","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Reinforcement; Conceptual framework; Process (computing); Heuristic; Interoperability","score_opus":0.07417532899007453,"score_gpt":0.34930266685967293,"score_spread":0.2751273378695984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3038500004","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003266661,0.46822307,0.48064142,0.011951479,0.0015427319,0.00009337827,0.000072088435,0.00019528152,0.034013957],"genre_scores_gemma":[0.30631495,0.48742235,0.18968557,0.0039148484,0.0025601184,0.0004459998,0.00015814348,0.00014295742,0.009354985],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985251,0.0005744262,0.000110429384,0.0003058037,0.00039699735,0.00008721607],"domain_scores_gemma":[0.99662566,0.002199419,0.00023187224,0.00020520968,0.00059049827,0.00014733341],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036096724,0.0014896543,0.0013807577,0.0014528946,0.00054054894,0.0038969005,0.004910993,0.0035794317,0.0023629328],"category_scores_gemma":[0.0047001923,0.00058706215,0.00094644376,0.0019441406,0.007223248,0.0045837597,0.0018051903,0.005024124,0.00085519755],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002981732,0.00009731337,0.00029664597,0.0017975396,0.000075583244,0.000069514506,0.0002469929,0.04065829,0.0004522047,0.8179275,0.0036571498,0.13469143],"study_design_scores_gemma":[0.000052119805,0.00024693052,0.00054542883,0.0025946007,0.00012792995,0.0002914418,0.00031415978,0.10869359,0.000974257,0.6732743,0.212773,0.0001122396],"about_ca_topic_score_codex":0.003901859,"about_ca_topic_score_gemma":0.0027164537,"teacher_disagreement_score":0.004910993,"about_ca_system_score_codex":0.0037944177,"about_ca_system_score_gemma":0.0034001037,"threshold_uncertainty_score":0.027530551},"labels":[],"label_agreement":null},{"id":"W3038915804","doi":"10.48550/arxiv.2007.02151","title":"Variational Policy Gradient Method for Reinforcement Learning with General Utilities","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Reinforcement; Computer science; Gradient method; Mathematical optimization; Mathematical economics; Artificial intelligence; Mathematics; Engineering; Structural engineering","score_opus":0.084019057565281,"score_gpt":0.22943373992515428,"score_spread":0.1454146823598733,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3038915804","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018843543,0.0002621269,0.99563545,0.00022941698,0.00003898012,0.000033163906,0.000021912625,0.00009167508,0.0018029639],"genre_scores_gemma":[0.41974533,0.0011504834,0.5580292,0.0005078691,0.00019941048,0.0007445747,0.000293626,0.000535463,0.018793996],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993222,0.00035244567,0.00002332786,0.00009570161,0.00014893347,0.000057459834],"domain_scores_gemma":[0.99838984,0.001205623,0.0000732881,0.000072125455,0.00018911032,0.00007001888],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022791633,0.0011555777,0.0013633001,0.00071382907,0.00041544152,0.0010641488,0.0013789429,0.0014971131,0.004098907],"category_scores_gemma":[0.006133439,0.0006659971,0.00083063653,0.00059603644,0.0016976625,0.0013812102,0.0017476968,0.002557037,0.00072791206],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000043449923,0.0000344514,0.00033076137,0.000102724276,0.0000420101,0.000064419386,0.00006727513,0.7629789,0.00085005333,0.2138607,0.0014564309,0.020168817],"study_design_scores_gemma":[0.0000056316458,0.0000061604346,0.000016689119,0.0000053916283,0.0000020783448,0.000004037673,0.0000028571762,0.97652197,0.000092858434,0.022754941,0.00058471377,0.0000026008288],"about_ca_topic_score_codex":0.006985894,"about_ca_topic_score_gemma":0.0044826525,"teacher_disagreement_score":0.006985894,"about_ca_system_score_codex":0.002092689,"about_ca_system_score_gemma":0.0022653188,"threshold_uncertainty_score":0.015183568},"labels":[],"label_agreement":null},{"id":"W3039671729","doi":"10.48550/arxiv.2007.02863","title":"Counterfactual Data Augmentation using Locally Factored Dynamics","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Counterfactual thinking; Reinforcement learning; Computer science; Coda; Representation (politics); Artificial intelligence; Set (abstract data type); Causal structure; Sequence (biology); Machine learning; State space; Mathematics","score_opus":0.2323526430308347,"score_gpt":0.23812963320514827,"score_spread":0.005776990174313573,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3039671729","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029106248,0.0001714764,0.96795803,0.00025468966,0.000046843106,0.000069201764,0.00010197595,0.0012405857,0.0010509865],"genre_scores_gemma":[0.7635284,0.000108107815,0.23439668,0.00017036367,0.00004610542,0.0001973157,0.0003253335,0.00013679643,0.0010909104],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99872273,0.00052382844,0.000064003514,0.0003633653,0.0002359171,0.00009004131],"domain_scores_gemma":[0.9924873,0.0045018177,0.00077034574,0.0014999907,0.00052955037,0.00021093912],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024856345,0.0011729515,0.0012400192,0.0006849431,0.00047602848,0.0011772116,0.0017239511,0.0011299718,0.0024394777],"category_scores_gemma":[0.01228546,0.0007530269,0.00094543525,0.0005573228,0.00198413,0.0033303893,0.002891295,0.0021817621,0.00039731688],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048135492,0.00019567845,0.0028118119,0.000137098,0.00010796295,0.00018056872,0.00030870433,0.8326565,0.0063968743,0.03137887,0.0014628557,0.123881735],"study_design_scores_gemma":[0.000013235994,0.000028376297,0.00009163161,0.0000074139293,0.000005423722,0.000016457278,0.00000830823,0.9868362,0.00095215446,0.011678146,0.00035553245,0.0000071787713],"about_ca_topic_score_codex":0.0036024048,"about_ca_topic_score_gemma":0.00414749,"teacher_disagreement_score":0.0036024048,"about_ca_system_score_codex":0.0010041159,"about_ca_system_score_gemma":0.0014569494,"threshold_uncertainty_score":0.013145447},"labels":[],"label_agreement":null},{"id":"W3040403450","doi":"10.48550/arxiv.2007.03151","title":"Curriculum learning for multilevel budgeted combinatorial problems","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Heuristics; Reinforcement learning; Solver; Combinatorial optimization; Mathematical optimization; Computer science; Speedup; Graph; Theoretical computer science; Mathematics; Artificial intelligence","score_opus":0.08203876657349847,"score_gpt":0.2036334493881823,"score_spread":0.12159468281468384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3040403450","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18714437,0.0012794349,0.7961976,0.0023691433,0.00015281007,0.0003370674,0.0010926089,0.0024015778,0.009025283],"genre_scores_gemma":[0.6959669,0.00042567297,0.29578018,0.00058389653,0.00010836012,0.0005928113,0.0018389589,0.00034626658,0.00435706],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990551,0.00037687153,0.000040897874,0.0002700743,0.00013077375,0.00012629297],"domain_scores_gemma":[0.9947673,0.0040855487,0.00031921832,0.00036469923,0.0002578515,0.00020538726],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015880127,0.0014537466,0.0016276458,0.0008198641,0.00059422187,0.0012768747,0.0022251776,0.002009203,0.0058534113],"category_scores_gemma":[0.010575292,0.00072909356,0.001095857,0.00094908505,0.0014932787,0.0025414205,0.0015309586,0.0033327865,0.00056966895],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011110014,0.00020953934,0.0016544996,0.00022173891,0.000049445884,0.00005678446,0.000056077533,0.9363659,0.00059213216,0.013457626,0.0034759464,0.043749195],"study_design_scores_gemma":[0.000030601288,0.000024836969,0.00011646656,0.000012009709,0.0000057740326,0.000006829561,0.000010425651,0.9818409,0.00023465695,0.017318897,0.00039578803,0.0000028749807],"about_ca_topic_score_codex":0.0051731756,"about_ca_topic_score_gemma":0.010254064,"teacher_disagreement_score":0.0058534113,"about_ca_system_score_codex":0.0019874158,"about_ca_system_score_gemma":0.0016983423,"threshold_uncertainty_score":0.019581616},"labels":[],"label_agreement":null},{"id":"W3040829104","doi":"10.48550/arxiv.2007.03749","title":"Sharp Analysis of Smoothed Bellman Error Embedding","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Reinforcement learning; Embedding; Bellman equation; Artificial neural network; Function (biology); Representation (politics); Horizon; Computer science; Nonlinear system; Algorithm; Mathematics; Discrete mathematics; Applied mathematics; Mathematical optimization; Physics; Artificial intelligence; Quantum mechanics; Geometry","score_opus":0.1156203499844863,"score_gpt":0.22891043717233997,"score_spread":0.11329008718785367,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3040829104","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039841156,0.00040456947,0.9534352,0.00073065143,0.00007074132,0.00004614478,0.00008840853,0.00043781532,0.0049452474],"genre_scores_gemma":[0.85933584,0.00039084002,0.12988606,0.00045993435,0.00009396894,0.00022765255,0.00023716885,0.0002957146,0.009072911],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9985176,0.0005588228,0.000056283763,0.0002829579,0.00040003896,0.00018432907],"domain_scores_gemma":[0.990276,0.0068408577,0.00069934863,0.0007235066,0.00097300624,0.0004873458],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033979523,0.0011866428,0.0011862015,0.00070016744,0.0005887247,0.001332897,0.0016422794,0.0018727685,0.004408114],"category_scores_gemma":[0.020841401,0.00064086035,0.0006742728,0.00047909695,0.00254709,0.0026884112,0.0028372665,0.0036061942,0.00059379335],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003216538,0.000070783004,0.0011266494,0.00019869652,0.000064353764,0.0001138465,0.0001677279,0.6555449,0.0050623687,0.30572724,0.003083868,0.02851792],"study_design_scores_gemma":[0.000011867651,0.000050011062,0.00011426126,0.000015364296,0.000004896303,0.000013414482,0.000006944212,0.95598435,0.00072140765,0.042707317,0.00036166052,0.000008582222],"about_ca_topic_score_codex":0.0022107288,"about_ca_topic_score_gemma":0.0016367829,"teacher_disagreement_score":0.004408114,"about_ca_system_score_codex":0.0019265641,"about_ca_system_score_gemma":0.0018395608,"threshold_uncertainty_score":0.017970324},"labels":[],"label_agreement":null},{"id":"W3040979139","doi":"10.48550/arxiv.2007.06184","title":"Efficient Planning in Large MDPs with Weak Linear Function Approximation","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Markov decision process; Bellman equation; Oracle; Mathematical optimization; Computer science; Computation; Function (biology); Set (abstract data type); Time horizon; State (computer science); Core (optical fiber); Approximation algorithm; Expected value; Markov process; Algorithm; Mathematics","score_opus":0.06112267459238138,"score_gpt":0.19371696036882993,"score_spread":0.13259428577644855,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3040979139","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026054002,0.00017249415,0.97106206,0.00031461983,0.000016615633,0.000054593325,0.0000767501,0.000881153,0.0013676209],"genre_scores_gemma":[0.71294284,0.00025450432,0.28350148,0.00019144549,0.000043073658,0.00035290897,0.0003378025,0.0002464361,0.0021295256],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989925,0.0003989766,0.000060386465,0.00022928327,0.00018223033,0.00013666123],"domain_scores_gemma":[0.9932486,0.0055836933,0.00034349406,0.0004521526,0.00019233205,0.0001798489],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021222723,0.0013549861,0.0014889969,0.00060722686,0.00067708705,0.0013117846,0.0016033791,0.0014372872,0.0019119894],"category_scores_gemma":[0.008949227,0.0010135816,0.0009527974,0.0008016557,0.002112797,0.002289736,0.0024549938,0.0024929042,0.0004809549],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000059598216,0.000022888846,0.0002743639,0.000040651135,0.000014244737,0.000028085004,0.000024880415,0.9812733,0.00039052754,0.007426738,0.00027324524,0.010171468],"study_design_scores_gemma":[0.000008864743,0.0000068934605,0.000019640283,0.0000026000505,0.0000019492188,0.0000030169313,0.0000041000653,0.98917854,0.00022030009,0.0104786595,0.00007405551,0.0000014037087],"about_ca_topic_score_codex":0.005495476,"about_ca_topic_score_gemma":0.0048843184,"teacher_disagreement_score":0.005495476,"about_ca_system_score_codex":0.0018617627,"about_ca_system_score_gemma":0.0018585826,"threshold_uncertainty_score":0.013508141},"labels":[],"label_agreement":null},{"id":"W3041568888","doi":"10.48550/arxiv.2007.06049","title":"An Equivalence between Loss Functions and Non-Uniform Sampling in Experience Replay","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Equivalence (formal languages); Reinforcement learning; Function (biology); Mathematics; Computer science; Algorithm; Applied mathematics; Statistics; Discrete mathematics; Artificial intelligence","score_opus":0.13083770781811202,"score_gpt":0.2445806165634153,"score_spread":0.11374290874530327,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3041568888","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029047927,0.00018842718,0.96822953,0.00036021014,0.00003614953,0.000055252814,0.00003456314,0.00030386675,0.001744114],"genre_scores_gemma":[0.84090704,0.00022004137,0.15498583,0.00032459956,0.00006744876,0.0002516364,0.000100372396,0.00014553253,0.0029974782],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9969656,0.0013858182,0.00015704313,0.0006322673,0.0006181087,0.00024125793],"domain_scores_gemma":[0.9928296,0.004283119,0.0007503356,0.0011494941,0.0006547752,0.00033272992],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00543351,0.0012360917,0.0011090328,0.0005308621,0.0005100612,0.0012946966,0.0022052494,0.0017526611,0.0017160861],"category_scores_gemma":[0.024222374,0.00053080235,0.00059136923,0.00049137394,0.0025742021,0.0042654485,0.0029960775,0.0032330824,0.0003593082],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004312453,0.00026206698,0.0021883615,0.00012187225,0.0000775058,0.00013384971,0.00020138398,0.80058444,0.005004506,0.10330052,0.0015824299,0.08611185],"study_design_scores_gemma":[0.000024286546,0.0001568242,0.00031545595,0.000013474189,0.0000079824,0.000041270236,0.000011284268,0.9676082,0.0014643735,0.029950505,0.00039501584,0.000011353764],"about_ca_topic_score_codex":0.00182994,"about_ca_topic_score_gemma":0.0013423803,"teacher_disagreement_score":0.00543351,"about_ca_system_score_codex":0.0015446042,"about_ca_system_score_gemma":0.0012273068,"threshold_uncertainty_score":0.028735518},"labels":[],"label_agreement":null},{"id":"W3041828093","doi":"10.24963/ijcai.2020/706","title":"On Overfitting and Asymptotic Bias in Batch Reinforcement Learning with Partial Observability (Extended Abstract)","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; Samsung; Institut de Valorisation des Données; Waalse Gewest","keywords":"Overfitting; Observability; Reinforcement learning; Term (time); Context (archaeology); Representation (politics); Computer science; Artificial intelligence; State (computer science); Mathematics; Applied mathematics; Algorithm; Artificial neural network","score_opus":0.04988424548801556,"score_gpt":0.2554322823185893,"score_spread":0.2055480368305737,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3041828093","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.052005198,0.00083080214,0.9414865,0.0014828265,0.00006227326,0.000038953895,0.00004460744,0.00027494074,0.003774004],"genre_scores_gemma":[0.9309776,0.00079631136,0.06468233,0.0007389377,0.00018600283,0.00015255417,0.00006779564,0.0001711807,0.002227392],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99586797,0.0019925039,0.00019593808,0.00052368204,0.0010622782,0.0003576793],"domain_scores_gemma":[0.9282853,0.063265845,0.003166557,0.0020804242,0.002592102,0.0006098261],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010635027,0.0009945376,0.0017191103,0.0007206368,0.00042990848,0.0014724972,0.001356987,0.0014942075,0.0020736174],"category_scores_gemma":[0.07398006,0.00059675553,0.00083622214,0.00074326963,0.0043937652,0.003087281,0.0029883888,0.0027947766,0.0001860811],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030160235,0.000116011346,0.0029325145,0.00026584504,0.00017592243,0.00022722219,0.00023902528,0.83555394,0.0026799487,0.11742869,0.00132302,0.038756363],"study_design_scores_gemma":[0.00003280029,0.00010522101,0.0006378588,0.000049118113,0.000028298602,0.000045948633,0.000017387041,0.8926876,0.0006583194,0.105457194,0.00025956758,0.000020568557],"about_ca_topic_score_codex":0.0042722286,"about_ca_topic_score_gemma":0.002491111,"teacher_disagreement_score":0.010635027,"about_ca_system_score_codex":0.0023654015,"about_ca_system_score_gemma":0.0016171802,"threshold_uncertainty_score":0.056244075},"labels":[],"label_agreement":null},{"id":"W3045293119","doi":"10.48550/arxiv.2007.10916","title":"On the Convergence of Reinforcement Learning with Monte Carlo Exploring Starts","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Convergence (economics); Mathematical optimization; Bellman equation; Monte Carlo method; Computer science; Complement (music); Markov decision process; Function (biology); Stochastic approximation; Applied mathematics; Mathematics; Markov process; Artificial intelligence; Economics","score_opus":0.1503308442699026,"score_gpt":0.18248930565484367,"score_spread":0.03215846138494108,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3045293119","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025146496,0.00083972415,0.96391135,0.0007570683,0.00006887343,0.00007648507,0.00006683035,0.0002242084,0.008908959],"genre_scores_gemma":[0.7940237,0.0017225107,0.19446638,0.00044643745,0.0001592897,0.0005721147,0.000279052,0.0004679432,0.007862472],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99641055,0.0021714005,0.00011387223,0.00041618978,0.00059893564,0.0002890827],"domain_scores_gemma":[0.94845355,0.045616593,0.0015878917,0.001158266,0.0021549335,0.0010287449],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008696639,0.0015748424,0.0023979065,0.0015087654,0.0009766119,0.0016398459,0.002400978,0.001898637,0.005304918],"category_scores_gemma":[0.057096563,0.0009830307,0.0013871626,0.000998235,0.004247706,0.0032306106,0.0032972863,0.0034646466,0.0006290004],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016852857,0.000049009406,0.001036587,0.00012298951,0.0000741606,0.000072020535,0.0001769369,0.78737247,0.00044354217,0.19856298,0.0008415439,0.011079143],"study_design_scores_gemma":[0.000019466976,0.000032287247,0.00009094881,0.00003799476,0.000009252586,0.000012672006,0.000013202551,0.92877257,0.00013888591,0.070519276,0.00034381414,0.000009567258],"about_ca_topic_score_codex":0.007153327,"about_ca_topic_score_gemma":0.003589896,"teacher_disagreement_score":0.008696639,"about_ca_system_score_codex":0.002551484,"about_ca_system_score_gemma":0.0027302864,"threshold_uncertainty_score":0.04599279},"labels":[],"label_agreement":null},{"id":"W3046590403","doi":"10.1016/j.ibmed.2024.100137","title":"Reinforcement learning in large, structured action spaces: A simulation study of decision support for spinal cord injury rehabilitation","year":2024,"lang":"en","type":"article","venue":"Intelligence-Based Medicine","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Parkwood Institute; Lawson Health Research Institute; Western University","funders":"","keywords":"Rehabilitation; Reinforcement; Reinforcement learning; Spinal cord injury; Action (physics); Physical medicine and rehabilitation; Psychology; Spinal cord; Computer science; Medicine; Artificial intelligence; Neuroscience; Social psychology","score_opus":0.05463180441145013,"score_gpt":0.39976203145658495,"score_spread":0.34513022704513485,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3046590403","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95132005,0.0004415355,0.03653832,0.001833078,0.00011549192,0.00032939756,0.0004052739,0.0001280959,0.008888846],"genre_scores_gemma":[0.98747265,0.00014243109,0.010308833,0.0001805839,0.000016424428,0.00023893108,0.00019851995,0.000014014743,0.001427495],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99862707,0.0009559862,0.000041755797,0.00012828078,0.000086865366,0.00016000803],"domain_scores_gemma":[0.9806082,0.01668392,0.0007014479,0.00037215638,0.00056578283,0.0010684823],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002431535,0.0008189521,0.00091044774,0.00056789455,0.0008302582,0.0009288268,0.001256533,0.0020849274,0.005091343],"category_scores_gemma":[0.013266664,0.0003625562,0.0011411568,0.00048323534,0.0014508174,0.0012616838,0.0014286144,0.0027542203,0.0002763726],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005049042,0.00092462584,0.005062378,0.000111565154,0.00010628265,0.00025866355,0.00024023313,0.9812185,0.00041063788,0.00542428,0.00088774663,0.0048501897],"study_design_scores_gemma":[0.00025346145,0.00046256473,0.0010334614,0.000021991842,0.000018936,0.000019745365,0.00015562914,0.99292725,0.00019899424,0.0043830033,0.00050859636,0.000016335967],"about_ca_topic_score_codex":0.02157766,"about_ca_topic_score_gemma":0.019304516,"teacher_disagreement_score":0.02157766,"about_ca_system_score_codex":0.0016675792,"about_ca_system_score_gemma":0.0018287565,"threshold_uncertainty_score":0.04290414},"labels":[],"label_agreement":null},{"id":"W3047253995","doi":"10.1109/lwc.2020.3045005","title":"Faded-Experience Trust Region Policy Optimization for Model-Free Power Allocation in Interference Channel","year":2020,"lang":"en","type":"preprint","venue":"IEEE Wireless Communications Letters","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Reinforcement learning; Memorization; Convergence (economics); Computer science; Interference (communication); Channel (broadcasting); Power (physics); Control (management); Artificial intelligence; Telecommunications; Mathematics; Economics","score_opus":0.07360881230310362,"score_gpt":0.30746428122084635,"score_spread":0.23385546891774273,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3047253995","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02408271,0.0003020497,0.9736241,0.0002135056,0.000030244963,0.000029378038,0.000020466427,0.00021622225,0.0014813378],"genre_scores_gemma":[0.95131695,0.0001897937,0.046552505,0.00008685139,0.000025856523,0.000100380516,0.000036953366,0.000048029164,0.0016426679],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994636,0.00022752161,0.00002368041,0.00009004187,0.0001041204,0.00009095654],"domain_scores_gemma":[0.9977036,0.0016398905,0.0002050793,0.0001062066,0.00024307084,0.000102167694],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014713339,0.0007947319,0.0012729252,0.0003459837,0.0002623211,0.000785879,0.0008513997,0.0008805999,0.0014000329],"category_scores_gemma":[0.006026149,0.00048318264,0.00049449765,0.00031725373,0.00120589,0.0008127189,0.0009933605,0.001347529,0.00020292639],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000040429637,0.000018428538,0.00019722001,0.00002844393,0.000013397959,0.000027690276,0.000029534502,0.98857343,0.00034968136,0.00418612,0.0002227491,0.0063129007],"study_design_scores_gemma":[0.0000034457883,0.000009778091,0.000014720977,0.0000015351523,0.000001390319,0.0000022399067,0.0000014776358,0.9987847,0.00007201567,0.0010681648,0.000039498755,0.0000010675125],"about_ca_topic_score_codex":0.0070128446,"about_ca_topic_score_gemma":0.0025605506,"teacher_disagreement_score":0.0070128446,"about_ca_system_score_codex":0.0009962289,"about_ca_system_score_gemma":0.0012977473,"threshold_uncertainty_score":0.01394403},"labels":[],"label_agreement":null},{"id":"W3072315125","doi":"10.1073/pnas.1907370117","title":"Fast reinforcement learning with generalized policy updates","year":2020,"lang":"en","type":"article","venue":"Proceedings of the National Academy of Sciences","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":82,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Leverage (statistics); Artificial intelligence; Machine learning; Generalization; Exploit; Mathematics","score_opus":0.03975041601643639,"score_gpt":0.28852758406207973,"score_spread":0.24877716804564334,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3072315125","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017796613,0.00037932806,0.9773014,0.00042991168,0.00009849174,0.000094779665,0.00004801386,0.0010861533,0.0027652676],"genre_scores_gemma":[0.84839535,0.00028645137,0.14668779,0.00038585084,0.00011396693,0.0003860947,0.00015126918,0.00014899031,0.003444324],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99892765,0.00041421357,0.000053267795,0.00019838402,0.00023982662,0.00016670693],"domain_scores_gemma":[0.9965654,0.0023774195,0.00025137485,0.00030391742,0.00033045214,0.0001713961],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027155834,0.0015539386,0.0018270706,0.00054125994,0.00043456836,0.0010256534,0.0017213156,0.0015717332,0.003563254],"category_scores_gemma":[0.009218163,0.0008093948,0.0005356335,0.000562767,0.0015915328,0.0017556439,0.0019573942,0.0024915102,0.0005995301],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001572232,0.000078547724,0.00063120975,0.00008673107,0.00005680752,0.00008720789,0.00005496436,0.9293624,0.0009385308,0.021051662,0.0017152629,0.045779467],"study_design_scores_gemma":[0.000029070969,0.000019795942,0.000034745248,0.000004446696,0.000004425606,0.000006427604,0.000002907001,0.99130976,0.00015882317,0.008206262,0.00022031307,0.0000031053094],"about_ca_topic_score_codex":0.006226778,"about_ca_topic_score_gemma":0.0065431264,"teacher_disagreement_score":0.006226778,"about_ca_system_score_codex":0.0012480393,"about_ca_system_score_gemma":0.0021096657,"threshold_uncertainty_score":0.01436156},"labels":[],"label_agreement":null},{"id":"W3080779222","doi":"10.48550/arxiv.2008.11329","title":"Inverse Policy Evaluation for Value-based Sequential Decision-making","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Value (mathematics); Inverse; Computer science; Operations research; Mathematics; Machine learning","score_opus":0.14822902897177312,"score_gpt":0.27112859961845015,"score_spread":0.12289957064667703,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3080779222","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008842974,0.00019746309,0.9871645,0.00024299655,0.000028176388,0.00008921094,0.000021305379,0.0001701631,0.0032432668],"genre_scores_gemma":[0.6648573,0.00027804443,0.33041033,0.0002306591,0.000057902726,0.00049530325,0.00010436556,0.00013654857,0.0034294373],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9962542,0.001929285,0.0001566032,0.0005208014,0.00088864955,0.0002503188],"domain_scores_gemma":[0.9895984,0.008611002,0.00044531515,0.00031025248,0.00074702175,0.00028800944],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006490166,0.0013015213,0.0019714902,0.00096982496,0.0007352536,0.0020515292,0.0015512768,0.0017759487,0.0034735696],"category_scores_gemma":[0.019129883,0.00085117534,0.0008948684,0.00076144666,0.003508955,0.0022384909,0.0019204426,0.0033242002,0.00046838878],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000103723025,0.000066317676,0.00043315435,0.00010135208,0.000040499926,0.000048113936,0.00013211709,0.88299847,0.00047534422,0.08355796,0.00045413538,0.031588886],"study_design_scores_gemma":[0.000012160098,0.000023216378,0.000024821791,0.0000129849595,0.0000038787844,0.000005341253,0.0000048519355,0.96023786,0.00024903365,0.03916143,0.0002594388,0.0000050935546],"about_ca_topic_score_codex":0.005410183,"about_ca_topic_score_gemma":0.0042689075,"teacher_disagreement_score":0.006490166,"about_ca_system_score_codex":0.0033390454,"about_ca_system_score_gemma":0.0036938584,"threshold_uncertainty_score":0.034323692},"labels":[],"label_agreement":null},{"id":"W3081310128","doi":"10.3390/electronics9091363","title":"A Survey of Multi-Task Deep Reinforcement Learning","year":2020,"lang":"en","type":"article","venue":"Electronics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":160,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Forgetting; Deep learning; Transfer of learning; Task (project management); Active learning (machine learning); Machine learning; Engineering","score_opus":0.030447156903129203,"score_gpt":0.26008836771618027,"score_spread":0.22964121081305106,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3081310128","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009830999,0.16745944,0.7975252,0.002021105,0.00051189086,0.00013686158,0.00019668788,0.0009007192,0.021416964],"genre_scores_gemma":[0.4763805,0.24431224,0.2540784,0.0017160452,0.0016112577,0.0005568177,0.00092503754,0.0003999414,0.020019742],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99934703,0.00016561482,0.00007665298,0.00015125691,0.00020496623,0.00005450429],"domain_scores_gemma":[0.9989974,0.0005849884,0.000059278784,0.00009096216,0.00021461603,0.00005276046],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014627193,0.0012902736,0.0013574486,0.0008440828,0.00029402497,0.0014578203,0.0017024623,0.0012977769,0.0033476062],"category_scores_gemma":[0.0031071703,0.0006225387,0.0007290186,0.0015649616,0.00056697486,0.0017538329,0.0011768226,0.0015512605,0.0009669823],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012695426,0.00017440502,0.0015691125,0.0013115052,0.00015887772,0.00008886593,0.00008758921,0.19245581,0.0012894729,0.03299874,0.0070370343,0.7627016],"study_design_scores_gemma":[0.000038207705,0.00029774045,0.0010661255,0.0005622506,0.00009063315,0.00021428584,0.00007100674,0.885215,0.0023662131,0.04919489,0.060824625,0.00005907372],"about_ca_topic_score_codex":0.0031326504,"about_ca_topic_score_gemma":0.0022871722,"teacher_disagreement_score":0.0033476062,"about_ca_system_score_codex":0.0011256235,"about_ca_system_score_gemma":0.0012584848,"threshold_uncertainty_score":0.011198878},"labels":[],"label_agreement":null},{"id":"W3083684068","doi":"10.2139/ssrn.4059578","title":"Visualizing the Loss Landscape of Actor Critic Methods with Applications in Inventory Optimization","year":2022,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; University of Waterloo","funders":"","keywords":"Computer science","score_opus":0.012929530973607517,"score_gpt":0.30265001198992164,"score_spread":0.28972048101631415,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3083684068","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1416447,0.0019528207,0.8284966,0.0025319492,0.00026987796,0.00008002555,0.00047677264,0.0023387354,0.022208588],"genre_scores_gemma":[0.8421837,0.0007125995,0.14945047,0.000175498,0.000067430155,0.00011040582,0.00028287742,0.00065023097,0.0063667153],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99977785,0.00011129333,0.000008013876,0.000026083942,0.00005333144,0.00002353393],"domain_scores_gemma":[0.99845946,0.0011109941,0.00008859874,0.000070560585,0.00017132517,0.000099103214],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00134015,0.0010270434,0.00070053886,0.0011505713,0.000490695,0.0017590901,0.0007489295,0.0012143105,0.005659278],"category_scores_gemma":[0.004884279,0.0004238108,0.00041315774,0.0007374651,0.00065607636,0.0011619146,0.00093325326,0.001404956,0.0003915191],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011890127,0.000060873557,0.00076196634,0.00008777119,0.000021924932,0.00007999908,0.00011989134,0.94707286,0.0017021641,0.02411566,0.0034855716,0.02237238],"study_design_scores_gemma":[0.000008526748,0.000011201082,0.00017623097,0.000009897108,0.0000020824316,0.000009159283,0.000016123606,0.98953027,0.00020954924,0.009549142,0.00047269894,0.0000052548057],"about_ca_topic_score_codex":0.005975539,"about_ca_topic_score_gemma":0.0051542586,"teacher_disagreement_score":0.005975539,"about_ca_system_score_codex":0.00092085876,"about_ca_system_score_gemma":0.000769576,"threshold_uncertainty_score":0.018932164},"labels":[],"label_agreement":null},{"id":"W3087873943","doi":"10.48550/arxiv.2009.11997","title":"Continual Model-Based Reinforcement Learning with Hypernetworks","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Task (project management); Computer science; Artificial intelligence; Machine learning; Control (management); Dynamics (music); Plan (archaeology); State (computer science); Robot; Engineering","score_opus":0.06297797963901723,"score_gpt":0.1813221165108317,"score_spread":0.11834413687181446,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3087873943","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0633035,0.0004671748,0.929757,0.00051145506,0.00008364416,0.00008391998,0.00019833884,0.0015987392,0.003996255],"genre_scores_gemma":[0.92233366,0.00017974155,0.07322354,0.00018865732,0.000045014134,0.00023501812,0.00024116132,0.00012278075,0.0034304112],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99943894,0.00020050125,0.000031107968,0.00014786406,0.000112422545,0.000069054826],"domain_scores_gemma":[0.99690175,0.0020866562,0.00022658096,0.00028862702,0.00032561694,0.00017072199],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014039102,0.0009802907,0.0010178065,0.0005796114,0.0004045943,0.00093399873,0.0022251573,0.0011037319,0.0037915152],"category_scores_gemma":[0.005758398,0.0007036854,0.0005745293,0.00044507984,0.0013608782,0.0023309516,0.0015252199,0.0023571206,0.00050027313],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000093307506,0.00005889367,0.0004471292,0.00003797518,0.00002870117,0.000042455376,0.000046906596,0.9661875,0.0005342229,0.0058774524,0.0007010835,0.025944319],"study_design_scores_gemma":[0.000008574354,0.000012691424,0.00002557506,0.0000028948004,0.000002456434,0.0000039967717,0.0000024896833,0.99604034,0.00013374137,0.003638785,0.00012584441,0.000002676301],"about_ca_topic_score_codex":0.0071179704,"about_ca_topic_score_gemma":0.0071147126,"teacher_disagreement_score":0.0071179704,"about_ca_system_score_codex":0.0014106218,"about_ca_system_score_gemma":0.0009920015,"threshold_uncertainty_score":0.014153063},"labels":[],"label_agreement":null},{"id":"W3089723243","doi":"","title":"Learning Intrinsic Rewards as a Bi-Level Optimization Problem","year":2020,"lang":"en","type":"article","venue":"Uncertainty in Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Intrinsic motivation; Mathematical optimization; Mathematics; Psychology; Social psychology","score_opus":0.06927971827994504,"score_gpt":0.29235963903827544,"score_spread":0.22307992075833039,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3089723243","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021368152,0.000311094,0.9715674,0.0010100041,0.00005280093,0.00005524568,0.000097249045,0.0001831777,0.005354892],"genre_scores_gemma":[0.72457635,0.00041939472,0.25656196,0.00049721595,0.00016654137,0.00055750366,0.00025693185,0.00021821812,0.016745811],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99847454,0.00059027213,0.00006842547,0.00030621194,0.0003546251,0.00020594694],"domain_scores_gemma":[0.9959875,0.0029452997,0.00026233165,0.0001900053,0.000333149,0.00028179822],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030176812,0.0009159143,0.0024887554,0.0007877587,0.00057630986,0.0031767252,0.0026469724,0.004620814,0.0069547566],"category_scores_gemma":[0.007924285,0.001120555,0.00093357573,0.0011176021,0.0020491837,0.0034778013,0.0035940243,0.0036708764,0.0008517234],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000082819926,0.000074572315,0.00038005575,0.00010673295,0.000061186336,0.000064935906,0.000060347436,0.8974233,0.00059394195,0.08479225,0.0013347473,0.015025098],"study_design_scores_gemma":[0.000012480877,0.00001771422,0.000045352986,0.000007193823,0.000005565926,0.0000055256146,0.000004455778,0.98019385,0.000079625366,0.019401558,0.00022225764,0.0000043916284],"about_ca_topic_score_codex":0.0028970342,"about_ca_topic_score_gemma":0.0026254796,"teacher_disagreement_score":0.0069547566,"about_ca_system_score_codex":0.0022645914,"about_ca_system_score_gemma":0.001886153,"threshold_uncertainty_score":0.023266017},"labels":[],"label_agreement":null},{"id":"W3090658167","doi":"10.1109/ijcnn48605.2020.9207473","title":"Automatic Policy Decomposition through Abstract State Space Dynamic Specialization","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Royal Military College of Canada","funders":"","keywords":"Reinforcement learning; Computer science; Bottleneck; State space; Artificial intelligence; State (computer science); Q-learning; Space (punctuation); Bellman equation; Decomposition; Function (biology); Macro; Action (physics); Machine learning; Mathematical optimization; Algorithm; Mathematics","score_opus":0.018410848227613713,"score_gpt":0.3002906061636799,"score_spread":0.28187975793606623,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3090658167","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019451907,0.00012905338,0.97742647,0.00017926107,0.000019210502,0.000034413657,0.000088266395,0.0010803048,0.0015911227],"genre_scores_gemma":[0.80271226,0.00023128989,0.19253226,0.00020303843,0.00003603238,0.00017956053,0.00046217305,0.00024578383,0.0033975213],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994795,0.00014203298,0.00003453462,0.00016082461,0.00009729514,0.00008574957],"domain_scores_gemma":[0.9991041,0.00046186638,0.00009319317,0.0001725273,0.000092000424,0.0000764496],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009430054,0.00069867464,0.0010372412,0.00057582284,0.00031874108,0.0009189371,0.0008778036,0.00078027823,0.0038957738],"category_scores_gemma":[0.0027293102,0.0005495491,0.00081855385,0.000525217,0.0010041007,0.0021221158,0.0021306844,0.0016624985,0.00057805155],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013647926,0.00006770481,0.00087865844,0.00008950051,0.000042417265,0.0001233011,0.0001581758,0.8238918,0.004634095,0.051794626,0.0021176913,0.11606547],"study_design_scores_gemma":[0.000004316615,0.0000071607387,0.00004431534,0.0000041712333,0.0000024601804,0.000006597311,0.0000048943893,0.97787136,0.00044596364,0.021264452,0.00034200607,0.0000022806132],"about_ca_topic_score_codex":0.0033278512,"about_ca_topic_score_gemma":0.003946904,"teacher_disagreement_score":0.0038957738,"about_ca_system_score_codex":0.001168733,"about_ca_system_score_gemma":0.0011737418,"threshold_uncertainty_score":0.013032615},"labels":[],"label_agreement":null},{"id":"W3090693621","doi":"10.1109/ccta41146.2020.9206397","title":"Reinforcement Learning in Deep Structured Teams: Initial Results with Finite and Infinite Valued Features","year":2020,"lang":"en","type":"article","venue":"2020 IEEE Conference on Control Technology and Applications (CCTA)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Markov chain; Convergence (economics); Variable-order Markov model; Linear model; Quadratic equation; Linear regression; Orthonormal basis; Reinforcement learning; State (computer science)","score_opus":0.012479337956139673,"score_gpt":0.24442315684276597,"score_spread":0.2319438188866263,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3090693621","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026296936,0.0029387393,0.9618707,0.0011265108,0.00014655071,0.000045871493,0.00006680197,0.00016504078,0.0073428415],"genre_scores_gemma":[0.8808637,0.0038857167,0.10486096,0.00042156212,0.00038075537,0.00022061412,0.00020154554,0.00014451225,0.00902068],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988967,0.00050367333,0.000050416464,0.00022484154,0.00018310666,0.00014121577],"domain_scores_gemma":[0.9903148,0.007693896,0.00043920078,0.00037758742,0.0007539648,0.00042049552],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036916935,0.0015270726,0.001864068,0.0006507582,0.0007209349,0.0013531073,0.0018003491,0.0021469155,0.0033907616],"category_scores_gemma":[0.017676346,0.00080707856,0.0013624992,0.0007162205,0.0026663963,0.003393581,0.0022111675,0.0040524267,0.0004769946],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001490287,0.00013970955,0.0009189272,0.0003014345,0.000086654785,0.00012809629,0.0002484372,0.70710206,0.0007836237,0.26094738,0.0019088957,0.02728577],"study_design_scores_gemma":[0.000012196411,0.00003616649,0.00006946175,0.000018696821,0.00000998997,0.000010893669,0.000014398663,0.93842494,0.00013601495,0.0608041,0.00045597745,0.00000713118],"about_ca_topic_score_codex":0.00512905,"about_ca_topic_score_gemma":0.0027608436,"teacher_disagreement_score":0.00512905,"about_ca_system_score_codex":0.0020484535,"about_ca_system_score_gemma":0.0011744944,"threshold_uncertainty_score":0.01952374},"labels":[],"label_agreement":null},{"id":"W3090903721","doi":"10.1109/ijcnn48605.2020.9207681","title":"Noisy Importance Sampling Actor-Critic: An Off-Policy Actor-Critic With Experience Replay","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Reinforcement learning; Computer science; Sample (material); Set (abstract data type); Truncation (statistics); Convergence (economics); Noise (video); Domain (mathematical analysis); Sampling (signal processing); Artificial intelligence; Machine learning; Mathematics; Computer vision","score_opus":0.05571278835542469,"score_gpt":0.31350621939963186,"score_spread":0.25779343104420716,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3090903721","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014587599,0.00028488768,0.9805876,0.00013116204,0.000114501985,0.00008725236,0.000023625438,0.0010091412,0.0031743096],"genre_scores_gemma":[0.7484974,0.00023587342,0.2451219,0.00029842043,0.000085147,0.0002461898,0.00012859891,0.00023530207,0.0051512024],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99887615,0.00039842672,0.000045368284,0.00019178883,0.00038276572,0.00010552638],"domain_scores_gemma":[0.9976292,0.0012610088,0.000203599,0.00027119095,0.0004929561,0.00014201118],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022255299,0.0011765694,0.0012150261,0.00038923504,0.00038154682,0.0009137483,0.0019628438,0.0010733416,0.0020364288],"category_scores_gemma":[0.0071653863,0.0005288276,0.00044733504,0.00031397006,0.0009905876,0.0008424645,0.0012973493,0.0018869971,0.00060367346],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026010443,0.00012827295,0.0010783846,0.0001272155,0.00008834471,0.00012132242,0.000094244635,0.87924296,0.0034822598,0.008179355,0.002513062,0.10468454],"study_design_scores_gemma":[0.000014999397,0.00003818679,0.000059269023,0.000005171975,0.00000605526,0.000015696776,0.000003436431,0.99758303,0.00067807105,0.0010653762,0.0005263594,0.0000043782643],"about_ca_topic_score_codex":0.004146389,"about_ca_topic_score_gemma":0.0044344766,"teacher_disagreement_score":0.004146389,"about_ca_system_score_codex":0.00068740034,"about_ca_system_score_gemma":0.0015706987,"threshold_uncertainty_score":0.011769891},"labels":[],"label_agreement":null},{"id":"W3092156990","doi":"10.1613/jair.1.12440","title":"Reward Machines: Exploiting Reward Function Structure in Reinforcement Learning","year":2022,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":172,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; University of Toronto; Government of Canada; Canadian Institute for Advanced Research; Vector Institute; Microsoft Research","keywords":"Reinforcement learning; Computer science; Exploit; Counterfactual thinking; Function (biology); Artificial intelligence; Finite-state machine; Machine learning; Algorithm; Psychology","score_opus":0.11135646365839093,"score_gpt":0.37456319042376984,"score_spread":0.26320672676537893,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3092156990","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004660038,0.000083340376,0.9937045,0.00012505492,0.000016769913,0.000035022433,0.000025299774,0.0005355071,0.0008144121],"genre_scores_gemma":[0.46039012,0.0002563169,0.53626007,0.00023039775,0.000057161713,0.00037578453,0.00010568738,0.0002951884,0.0020293863],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99832875,0.0009114333,0.00007595006,0.00026667985,0.0003147154,0.000102588776],"domain_scores_gemma":[0.99436694,0.004296996,0.00036621792,0.00060422654,0.00023191313,0.00013379952],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028778268,0.00085950684,0.00084210734,0.00050753955,0.00042145877,0.0013702131,0.0014309264,0.0012633456,0.003297319],"category_scores_gemma":[0.011167706,0.0005888832,0.0008212834,0.0004891472,0.0027385745,0.0028015117,0.001508486,0.0027153548,0.00056111097],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002764249,0.0001254525,0.0010268751,0.0001839499,0.000060965758,0.00014881467,0.00028181385,0.61844987,0.005538596,0.27920255,0.0014100142,0.09329476],"study_design_scores_gemma":[0.000028042406,0.00005976033,0.00006505713,0.000021201784,0.000009422978,0.000022670678,0.000007260441,0.896973,0.0020857381,0.099312134,0.0014023357,0.000013287377],"about_ca_topic_score_codex":0.0014488079,"about_ca_topic_score_gemma":0.0013539484,"teacher_disagreement_score":0.003297319,"about_ca_system_score_codex":0.0010353817,"about_ca_system_score_gemma":0.0011840691,"threshold_uncertainty_score":0.015219629},"labels":[],"label_agreement":null},{"id":"W3092954297","doi":"10.48550/arxiv.2010.09163","title":"D2RL: Deep Dense Architectures in Reinforcement Learning","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Suite; Computer science; Artificial intelligence; Benchmark (surveying); Deep learning; Architecture; Variety (cybernetics); Artificial neural network; Reinforcement; Unsupervised learning; Generative grammar; Machine learning; Human–computer interaction; Engineering","score_opus":0.06018814793559971,"score_gpt":0.19460360121575873,"score_spread":0.13441545328015903,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3092954297","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009602013,0.00069718354,0.9810338,0.0006145967,0.00010191897,0.000057995196,0.0002287991,0.00311307,0.004550686],"genre_scores_gemma":[0.5825522,0.0007715725,0.40508288,0.00067149114,0.00010597297,0.0006115416,0.00083696796,0.0008474163,0.008520094],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994999,0.00017809514,0.000025035795,0.00011245052,0.00013088927,0.00005360089],"domain_scores_gemma":[0.9988471,0.00067963253,0.00008899505,0.00016609271,0.00012750251,0.000090663394],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001503603,0.00095759175,0.0007879137,0.0003734519,0.0003325082,0.001005163,0.0018886575,0.0013634644,0.006266187],"category_scores_gemma":[0.004857422,0.0005919402,0.00048127957,0.00042387686,0.0011143766,0.0014968893,0.0020900706,0.0027839574,0.001367038],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000138298,0.000106868916,0.00082359364,0.00019392476,0.00006436774,0.0000844787,0.00006609621,0.835149,0.0024382372,0.048710804,0.008565383,0.103659004],"study_design_scores_gemma":[0.000026525275,0.000025560003,0.00004736585,0.000012486542,0.0000045474358,0.000010871696,0.0000034296133,0.97304916,0.0006951495,0.024497977,0.0016220491,0.0000049582086],"about_ca_topic_score_codex":0.004682967,"about_ca_topic_score_gemma":0.0060363477,"teacher_disagreement_score":0.006266187,"about_ca_system_score_codex":0.0012404213,"about_ca_system_score_gemma":0.0013226287,"threshold_uncertainty_score":0.020962477},"labels":[],"label_agreement":null},{"id":"W3093511015","doi":"10.3390/s20215991","title":"Policy-Gradient and Actor-Critic Based State Representation Learning for Safe Driving of Autonomous Vehicles","year":2020,"lang":"en","type":"article","venue":"Sensors","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Autoencoder; Computer science; Representation (politics); Artificial intelligence; Reinforcement learning; Perception; Scheme (mathematics); Object (grammar); Function (biology); Deep learning; Machine learning; Simulation; Mathematics; Psychology; Law","score_opus":0.02501525026013139,"score_gpt":0.2744572081329379,"score_spread":0.24944195787280649,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3093511015","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017363431,0.0001983119,0.98031473,0.00020176632,0.000042258933,0.000025821973,0.000024244964,0.0004875186,0.0013419504],"genre_scores_gemma":[0.92115957,0.0001176712,0.07542698,0.00012270366,0.000029867188,0.00008359972,0.000079164434,0.00006764691,0.0029128222],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99963105,0.00010443641,0.00001612687,0.000101230005,0.00008642475,0.00006077569],"domain_scores_gemma":[0.9993511,0.00029938426,0.00009253606,0.00005277813,0.0001492943,0.00005486665],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00090862514,0.0007567728,0.00073190575,0.00033734366,0.00027297347,0.00069178874,0.001026065,0.00089644076,0.0011568514],"category_scores_gemma":[0.0022956282,0.0005027005,0.0004257413,0.00027060116,0.0007911104,0.0008641977,0.0008799076,0.0014094663,0.00027348037],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003425473,0.000030647494,0.00041364113,0.000027322416,0.00002000087,0.000039073107,0.000040297935,0.9680609,0.0015613043,0.004843573,0.0004498851,0.024479112],"study_design_scores_gemma":[0.000001702205,0.000009291646,0.000028795175,0.0000012566004,0.0000011497283,0.0000029026926,0.0000015882162,0.9988568,0.00017305418,0.0008363866,0.00008538989,0.0000015655045],"about_ca_topic_score_codex":0.006227026,"about_ca_topic_score_gemma":0.004945404,"teacher_disagreement_score":0.006227026,"about_ca_system_score_codex":0.00086655305,"about_ca_system_score_gemma":0.0014442173,"threshold_uncertainty_score":0.012381554},"labels":[],"label_agreement":null},{"id":"W3093827687","doi":"10.22215/etd/2016-11612","title":"Multi-Robot Learning in the Guarding a Territory Game","year":2016,"lang":"en","type":"dissertation","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Guard (computer science); Reinforcement learning; Computer science; A priori and a posteriori; Robot; Artificial intelligence; Game theory; Human–computer interaction; Mathematical economics; Mathematics; Epistemology","score_opus":0.02150014991520405,"score_gpt":0.2836851225046714,"score_spread":0.2621849725894674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3093827687","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4446564,0.00028035225,0.54194254,0.00074613234,0.000038778184,0.00011378602,0.000033813616,0.00015053197,0.012037706],"genre_scores_gemma":[0.97480994,0.00007807109,0.02254699,0.000049463844,0.000012848868,0.000071691065,0.0000139285585,0.000012975282,0.002404031],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994485,0.0002863963,0.000016764534,0.00009505694,0.000074097086,0.00007916185],"domain_scores_gemma":[0.99826473,0.0012573855,0.00016426704,0.000055349396,0.00007860129,0.00017972275],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011250923,0.0006553081,0.00068881875,0.00021056521,0.00037134805,0.00059178873,0.0008389417,0.0008912048,0.0016379927],"category_scores_gemma":[0.0036268928,0.00023320723,0.0004080855,0.00016505554,0.0013276138,0.0010938034,0.0010426651,0.0011332807,0.00015763455],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022859979,0.00013091264,0.0012280981,0.00006070708,0.00004522852,0.00018987081,0.00017142785,0.9518233,0.0018155194,0.030304814,0.00031259388,0.01368903],"study_design_scores_gemma":[0.000020422953,0.000061039915,0.0001348864,0.0000035100059,0.000003914007,0.000012552385,0.000018654067,0.991516,0.00021014063,0.007861824,0.00015319025,0.0000038099738],"about_ca_topic_score_codex":0.0031619035,"about_ca_topic_score_gemma":0.0023828673,"teacher_disagreement_score":0.0031619035,"about_ca_system_score_codex":0.0007593239,"about_ca_system_score_gemma":0.0006884125,"threshold_uncertainty_score":0.0062869787},"labels":[],"label_agreement":null},{"id":"W3093967600","doi":"10.48550/arxiv.2010.09890","title":"Watch-And-Help: A Challenge for Social Perception and Human-AI Collaboration","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Perception; Psychology; Data science; Sociology; Computer science","score_opus":0.11056369047256379,"score_gpt":0.24066389574610228,"score_spread":0.1301002052735385,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3093967600","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5701495,0.0048595825,0.3667884,0.013500356,0.0013825104,0.0015074598,0.003721454,0.0057506245,0.032340165],"genre_scores_gemma":[0.8659854,0.00058853306,0.12404154,0.0012349844,0.00015016433,0.00073338934,0.0030560545,0.0003494195,0.0038605297],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99199533,0.005272631,0.0002591949,0.0010652298,0.0011301197,0.00027741655],"domain_scores_gemma":[0.9811127,0.011426069,0.0009958366,0.0032500427,0.0012789861,0.0019364998],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054410785,0.0012049582,0.000954688,0.00048575783,0.001225068,0.0021351948,0.0020051484,0.0029758147,0.003410958],"category_scores_gemma":[0.028373523,0.00039475132,0.0007297769,0.00038548934,0.0026778537,0.0047551277,0.0046597295,0.0027208882,0.0013036105],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031030062,0.0051116594,0.029865684,0.003453208,0.00067762507,0.0007959211,0.005651787,0.28427753,0.026180957,0.057671275,0.09276009,0.49045128],"study_design_scores_gemma":[0.00050862465,0.0035515798,0.017792335,0.00032663892,0.00010663934,0.00080795353,0.0044169514,0.71277016,0.021705115,0.15280643,0.08490763,0.00029981622],"about_ca_topic_score_codex":0.0041868757,"about_ca_topic_score_gemma":0.0045188596,"teacher_disagreement_score":0.0054410785,"about_ca_system_score_codex":0.0012294627,"about_ca_system_score_gemma":0.0014160203,"threshold_uncertainty_score":0.028775573},"labels":[],"label_agreement":null},{"id":"W3093980962","doi":"","title":"Incremental Policy Gradients for Online Reinforcement Learning Control","year":2021,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Term (time); Control (management); Artificial intelligence; Econometrics; Machine learning; Economics","score_opus":0.01937927747768515,"score_gpt":0.280628564126172,"score_spread":0.26124928664848684,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3093980962","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032015448,0.0005371151,0.99310523,0.00018663229,0.000066356115,0.000043113818,0.000037765618,0.00056651916,0.002255655],"genre_scores_gemma":[0.66357225,0.0011011881,0.3262515,0.00034239824,0.00021323394,0.00051325926,0.00020380871,0.00040050576,0.007401835],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993568,0.00022041133,0.000037856058,0.000107316686,0.00021642953,0.00006117328],"domain_scores_gemma":[0.99806637,0.0013523733,0.00013736334,0.00011953937,0.00024621794,0.000078134675],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016156181,0.0012127934,0.00121595,0.00076461176,0.0003856845,0.0012274126,0.0014577772,0.0011426809,0.0038238696],"category_scores_gemma":[0.008275854,0.00054705725,0.0005416449,0.00061026827,0.0012028813,0.0015021046,0.0012645797,0.0023987303,0.0007245602],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000091673566,0.000066714485,0.00045284585,0.00016447557,0.00004164793,0.000055483066,0.000067214074,0.841387,0.00091758417,0.07705104,0.0023292876,0.07737507],"study_design_scores_gemma":[0.000008632725,0.000017213506,0.00003752719,0.000010526742,0.0000036139152,0.00000799464,0.0000022171582,0.97708386,0.00023944135,0.021873936,0.0007098595,0.0000051837396],"about_ca_topic_score_codex":0.0045337444,"about_ca_topic_score_gemma":0.0034783795,"teacher_disagreement_score":0.0045337444,"about_ca_system_score_codex":0.0014899933,"about_ca_system_score_gemma":0.0014528822,"threshold_uncertainty_score":0.01279217},"labels":[],"label_agreement":null},{"id":"W3097732461","doi":"10.48550/arxiv.2011.02565","title":"Diversity-Enriched Option-Critic","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Diversity (politics); Political science; Law","score_opus":0.1188464807474624,"score_gpt":0.19455811583723112,"score_spread":0.07571163508976872,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3097732461","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033527248,0.0002363248,0.9610822,0.00026996064,0.000044092445,0.0000494178,0.00007269476,0.00047657616,0.0042414777],"genre_scores_gemma":[0.88332015,0.000121624,0.111734286,0.00014152697,0.00003041606,0.00013246755,0.00012281099,0.000101345984,0.004295251],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994325,0.00020775241,0.000026004109,0.00011950555,0.00014322721,0.00007107476],"domain_scores_gemma":[0.998346,0.0010046831,0.00016008447,0.00015586572,0.00019657104,0.00013680925],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014545497,0.0009361039,0.00095703173,0.00044356866,0.00038742367,0.00094775576,0.0013989239,0.0011974644,0.0024301035],"category_scores_gemma":[0.004108062,0.0004732727,0.0006548542,0.00037091717,0.0011922999,0.0010731682,0.0014858765,0.0019079254,0.0004114151],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009950278,0.00003279821,0.00069023645,0.00004869751,0.000032110536,0.000090329486,0.000051239527,0.94994605,0.0019179446,0.018740349,0.00076848484,0.027582182],"study_design_scores_gemma":[0.00000909896,0.000020977453,0.00004968667,0.0000056876834,0.0000047861718,0.000016031612,0.000002845674,0.99241674,0.00042186052,0.0067606107,0.00028694788,0.0000047462445],"about_ca_topic_score_codex":0.0017514252,"about_ca_topic_score_gemma":0.0023370164,"teacher_disagreement_score":0.0024301035,"about_ca_system_score_codex":0.0009936988,"about_ca_system_score_gemma":0.0011873246,"threshold_uncertainty_score":0.008129537},"labels":[],"label_agreement":null},{"id":"W3098974658","doi":"","title":"Promoting Coordination through Policy Regularization in Multi-Agent Deep Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Polytechnique Montréal; Université de Montréal","funders":"","keywords":"Reinforcement learning; Computer science; Regularization (linguistics); Predictability; Action selection; Artificial intelligence; Machine learning; Action (physics); Mathematics","score_opus":0.014284941918207643,"score_gpt":0.2518850997686436,"score_spread":0.23760015785043595,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3098974658","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.059011597,0.00027674326,0.9377426,0.00051504734,0.00004855321,0.00006192238,0.0000235427,0.0006930861,0.0016269239],"genre_scores_gemma":[0.9184997,0.000097307166,0.07955124,0.00024135051,0.000037530106,0.00013717564,0.00004109947,0.00008310449,0.0013114007],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99914217,0.00042340741,0.000038438735,0.00014793537,0.00014213772,0.00010593174],"domain_scores_gemma":[0.9963097,0.0023677,0.00048944383,0.00031755547,0.0002934996,0.00022209066],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028143695,0.001016569,0.0010049399,0.00040248447,0.000512996,0.0007594129,0.0014612783,0.0012536544,0.0009503876],"category_scores_gemma":[0.009064031,0.00054704736,0.0003467868,0.00031610762,0.0016403673,0.0012055833,0.001494813,0.0019621507,0.00022636508],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006886323,0.00008461949,0.0010431226,0.00003035809,0.00003446238,0.000037636684,0.00006718428,0.96945715,0.0013057363,0.0065493416,0.00059404783,0.020727394],"study_design_scores_gemma":[0.000009121997,0.000017477783,0.000038131835,0.0000030683286,0.0000020488148,0.000003390615,0.0000034039567,0.9976305,0.00019494336,0.0020090346,0.00008668967,0.0000022171273],"about_ca_topic_score_codex":0.0038875681,"about_ca_topic_score_gemma":0.0046441043,"teacher_disagreement_score":0.0038875681,"about_ca_system_score_codex":0.0011103606,"about_ca_system_score_gemma":0.0015018765,"threshold_uncertainty_score":0.014883995},"labels":[],"label_agreement":null},{"id":"W3099478911","doi":"10.1007/978-3-030-61577-2_13","title":"Experiments and Applications of Emotional Human-Robot Interaction Systems","year":2020,"lang":"en","type":"book-chapter","venue":"Studies in computational intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Robot; Human–robot interaction; Human–computer interaction; Computer science; Process (computing); Natural (archaeology); Artificial intelligence; Social robot; Scheme (mathematics); Psychology; Robot control; Mobile robot; Mathematics","score_opus":0.1618150848356296,"score_gpt":0.3963985655761806,"score_spread":0.234583480740551,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3099478911","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7289251,0.0020589293,0.19903035,0.00094134215,0.00054998195,0.00085352577,0.000660972,0.0008915849,0.0660882],"genre_scores_gemma":[0.9596675,0.00048162995,0.03220405,0.0001224293,0.000032635657,0.00042350302,0.00023836383,0.000095766416,0.00673428],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99914324,0.0004716236,0.000050435065,0.00010650123,0.00016504737,0.00006319377],"domain_scores_gemma":[0.9985287,0.00094251387,0.00004137994,0.00026308876,0.00014533367,0.000078948004],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010932481,0.00040367467,0.0003258469,0.0002574658,0.0005230722,0.0007178991,0.0008459263,0.0006122882,0.009069455],"category_scores_gemma":[0.0045037684,0.00021482773,0.00023418562,0.0003108226,0.00092811696,0.0010326077,0.0010806961,0.0005695188,0.0010309686],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005453012,0.005696594,0.0037865923,0.0017808649,0.00019883794,0.00073405664,0.0039684433,0.11514333,0.43154886,0.07612343,0.0148543,0.34071174],"study_design_scores_gemma":[0.0011474333,0.011308275,0.017388701,0.00023584712,0.00022741473,0.0010908784,0.0027422784,0.45166335,0.3498071,0.115593836,0.048547775,0.00024699155],"about_ca_topic_score_codex":0.00041874917,"about_ca_topic_score_gemma":0.000257003,"teacher_disagreement_score":0.009069455,"about_ca_system_score_codex":0.0003327673,"about_ca_system_score_gemma":0.00015517209,"threshold_uncertainty_score":0.030340314},"labels":[],"label_agreement":null},{"id":"W3100366369","doi":"10.1561/2200000071","title":"An Introduction to Deep Reinforcement Learning","year":2018,"lang":"en","type":"article","venue":"Foundations and Trends® in Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1251,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Artificial intelligence; Computer science; Deep learning; Generalization; Field (mathematics); Robotics; Machine learning; Robot; Mathematics","score_opus":0.013791966264077308,"score_gpt":0.29059655087526653,"score_spread":0.27680458461118923,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3100366369","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010461361,0.03745495,0.9141513,0.005073014,0.0015776712,0.00006732728,0.0007104234,0.00077034876,0.039148785],"genre_scores_gemma":[0.16272369,0.113412954,0.63076204,0.0059399446,0.006087467,0.00077734346,0.0019766372,0.00073232205,0.07758758],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9995043,0.000110160196,0.000044377437,0.00011050068,0.00019041455,0.000040221264],"domain_scores_gemma":[0.99898416,0.00070604833,0.00004541449,0.000069934795,0.00014506369,0.000049439568],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008229304,0.0008366553,0.0006438546,0.000698091,0.00024864983,0.0013292992,0.0009674388,0.0014603289,0.012112248],"category_scores_gemma":[0.00271042,0.00050860964,0.0007122797,0.0011244966,0.0009260098,0.0015469968,0.0009403393,0.0037458956,0.0037669989],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000045000474,0.00009777046,0.0007419307,0.00082113827,0.00008142079,0.00018340023,0.00015111653,0.067481324,0.001948348,0.53431755,0.059899785,0.33423132],"study_design_scores_gemma":[0.000019134897,0.00007830806,0.00043546254,0.00045744405,0.000023475217,0.00027279308,0.000024793422,0.10922107,0.00090861204,0.57342845,0.3150787,0.00005175703],"about_ca_topic_score_codex":0.001815033,"about_ca_topic_score_gemma":0.001585789,"teacher_disagreement_score":0.012112248,"about_ca_system_score_codex":0.0011422335,"about_ca_system_score_gemma":0.00094103755,"threshold_uncertainty_score":0.040519536},"labels":[],"label_agreement":null},{"id":"W3101152497","doi":"","title":"On Efficiency in Hierarchical Reinforcement Learning","year":2020,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Machine learning","score_opus":0.02283141343365988,"score_gpt":0.25049009276771156,"score_spread":0.22765867933405168,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3101152497","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038023945,0.0017351776,0.9415618,0.0017268354,0.00009915572,0.000070813745,0.000087975626,0.00022091098,0.01647336],"genre_scores_gemma":[0.87390316,0.0015260418,0.11274391,0.00055548147,0.00025512054,0.0002648946,0.00015481225,0.00036356534,0.010233071],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99581903,0.0023752411,0.00019691938,0.00041202846,0.0008037924,0.00039294342],"domain_scores_gemma":[0.9597682,0.035490714,0.00081590854,0.0020087673,0.0014211003,0.00049537834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007894389,0.0012776112,0.0023150877,0.0010866853,0.0007262702,0.0021797607,0.002690543,0.0016859977,0.006612667],"category_scores_gemma":[0.038810004,0.0008834267,0.0009631118,0.0015138944,0.0037000757,0.006300599,0.003283378,0.0036849158,0.0006387731],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019525428,0.000110378016,0.0009029418,0.00017814526,0.00007878063,0.000047691625,0.00019085477,0.32775056,0.0010859788,0.62421376,0.0021228222,0.04312288],"study_design_scores_gemma":[0.000041073537,0.000042660584,0.00026475263,0.000023911714,0.00002328696,0.000017090593,0.000020454314,0.5754522,0.00054049725,0.4228132,0.0007499947,0.00001086709],"about_ca_topic_score_codex":0.0037456502,"about_ca_topic_score_gemma":0.0022708175,"teacher_disagreement_score":0.007894389,"about_ca_system_score_codex":0.003089639,"about_ca_system_score_gemma":0.001498281,"threshold_uncertainty_score":0.041749954},"labels":[],"label_agreement":null},{"id":"W3101675824","doi":"","title":"CoinDICE: Off-Policy Confidence Interval Estimation","year":2020,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Confidence interval; Reinforcement learning; Computer science; Embedding; Bellman equation; Mathematical optimization; Function (biology); Confidence region; Algorithm; Mathematics; Applied mathematics; Statistics; Artificial intelligence","score_opus":0.07032887715500136,"score_gpt":0.20192292744749268,"score_spread":0.13159405029249133,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3101675824","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011688642,0.0004268807,0.985535,0.00028182927,0.00004567411,0.000074652045,0.00007320975,0.0008015578,0.0010725416],"genre_scores_gemma":[0.61332965,0.0003177189,0.3829932,0.0004155631,0.00010370809,0.0003534736,0.0004508888,0.00046732172,0.001568555],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99086094,0.0043550604,0.00044445088,0.0015508304,0.0022236812,0.0005649171],"domain_scores_gemma":[0.9097375,0.07493442,0.004507284,0.0054421006,0.0041173142,0.0012613739],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0161446,0.0017003703,0.0031826713,0.0016799899,0.0006244658,0.003216287,0.004584654,0.0028869114,0.0035314283],"category_scores_gemma":[0.11503129,0.0012122014,0.001000299,0.0013817422,0.002846551,0.004399699,0.0048720757,0.006036969,0.0006443906],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000546149,0.00024640214,0.004323005,0.0002653845,0.00018854867,0.00015597818,0.00019947211,0.79034,0.0012739768,0.059420865,0.002535724,0.14050451],"study_design_scores_gemma":[0.00002788709,0.000054706812,0.00016840505,0.000028004462,0.000009630941,0.00003086587,0.000008603024,0.98179996,0.00097074424,0.016548,0.0003400541,0.000013073134],"about_ca_topic_score_codex":0.0033585473,"about_ca_topic_score_gemma":0.0023465601,"teacher_disagreement_score":0.0161446,"about_ca_system_score_codex":0.0020933563,"about_ca_system_score_gemma":0.0028310795,"threshold_uncertainty_score":0.085381866},"labels":[],"label_agreement":null},{"id":"W3102849613","doi":"10.48550/arxiv.2011.03125","title":"LBGP: Learning Based Goal Planning for Autonomous Following in Front","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Trajectory; Planner; Computer science; Robot; Reinforcement learning; Artificial intelligence; Front (military); Deep learning; Human–computer interaction; Simulation; Engineering","score_opus":0.08554689636773007,"score_gpt":0.2142208071226572,"score_spread":0.12867391075492712,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3102849613","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018488191,0.0001759432,0.9719678,0.00019078265,0.00007040753,0.00008082553,0.00010671393,0.006008129,0.0029112238],"genre_scores_gemma":[0.6654155,0.00017595461,0.3258653,0.00025504356,0.000033343622,0.00018687325,0.00034991468,0.00033169438,0.007386369],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998373,0.000034517972,0.0000072713374,0.00005160152,0.00003720942,0.000032094027],"domain_scores_gemma":[0.9997509,0.000090606714,0.000024626346,0.000050698058,0.000043975127,0.000039104754],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00051654055,0.0007020517,0.0005652162,0.00024001708,0.0003410381,0.00052152295,0.0015623374,0.0011021848,0.0040115984],"category_scores_gemma":[0.0011057757,0.00042075748,0.0003853844,0.00021121328,0.00071842776,0.0008984553,0.00136987,0.001391079,0.0013193658],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035122104,0.00032916738,0.0014275102,0.00015614797,0.00006309339,0.0002800061,0.0001802986,0.70300716,0.012351685,0.011462886,0.010079921,0.26031095],"study_design_scores_gemma":[0.000013677643,0.000041965817,0.00008190073,0.0000061921546,0.000003909528,0.00001949845,0.000007284211,0.9940546,0.00146233,0.0032456655,0.0010586171,0.000004313111],"about_ca_topic_score_codex":0.0070854975,"about_ca_topic_score_gemma":0.0067676613,"teacher_disagreement_score":0.0070854975,"about_ca_system_score_codex":0.0006080304,"about_ca_system_score_gemma":0.0011747417,"threshold_uncertainty_score":0.0140885115},"labels":[],"label_agreement":null},{"id":"W3103290077","doi":"","title":"Learning Agent Representations for Ice Hockey","year":2020,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Simon Fraser University","funders":"","keywords":"Ice hockey; Computer science; Artificial intelligence; Physical medicine and rehabilitation","score_opus":0.04542303430578692,"score_gpt":0.2864625706532663,"score_spread":0.24103953634747938,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3103290077","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16573109,0.00040452604,0.81182617,0.0007131636,0.00024643925,0.00011369774,0.0005510433,0.0013236048,0.019090263],"genre_scores_gemma":[0.9506051,0.00012498898,0.041662958,0.000070559196,0.000024359784,0.00006674246,0.0003126923,0.00007390536,0.0070586056],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99992657,0.000017110833,0.0000031411596,0.000021250999,0.000011567197,0.00002036249],"domain_scores_gemma":[0.99978656,0.000093557384,0.000023929491,0.000028498418,0.000042500156,0.00002493849],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00021333194,0.00037676858,0.00054136134,0.00025923463,0.00030217637,0.00075107405,0.0007674174,0.00076686183,0.0049221464],"category_scores_gemma":[0.0014185436,0.00025281217,0.0002809257,0.00023795656,0.00041772155,0.001136668,0.00066442386,0.0008795731,0.00047925295],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001244157,0.000049785147,0.00064904493,0.00003516791,0.00002545994,0.000063840715,0.00005396103,0.93397754,0.00083007145,0.02007379,0.0030999924,0.041017003],"study_design_scores_gemma":[0.00000626071,0.000008159613,0.00004260735,0.000002991671,0.0000026380212,0.0000041490384,0.000008885026,0.99278635,0.00016717764,0.0066406145,0.00032837814,0.0000017438583],"about_ca_topic_score_codex":0.0075364155,"about_ca_topic_score_gemma":0.0102971615,"teacher_disagreement_score":0.0075364155,"about_ca_system_score_codex":0.0006839917,"about_ca_system_score_gemma":0.0005756025,"threshold_uncertainty_score":0.01646626},"labels":[],"label_agreement":null},{"id":"W3103898737","doi":"10.1109/tro.2022.3192969","title":"Joint Estimation of Expertise and Reward Preferences From Human Demonstrations","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Robotics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Inference; Robot; Leverage (statistics); Computer science; Human–robot interaction; Artificial intelligence; Set (abstract data type); Function (biology); Machine learning; Human–computer interaction; Human behavior; Space (punctuation)","score_opus":0.036868723667860044,"score_gpt":0.2569113578331829,"score_spread":0.22004263416532288,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3103898737","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.36854568,0.00035382135,0.6276942,0.00035290624,0.000014612102,0.000080898615,0.00016344212,0.00043473108,0.0023597348],"genre_scores_gemma":[0.9704756,0.00006282362,0.028761363,0.000036758276,0.00000841868,0.000029857798,0.00007953762,0.00001726459,0.00052840676],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99877506,0.0005659384,0.000055479235,0.0002957035,0.00020640322,0.000101289756],"domain_scores_gemma":[0.9908704,0.0066816485,0.0009455423,0.0006278549,0.00048356943,0.00039093685],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025095732,0.0005807838,0.0009082872,0.00058825716,0.00020602882,0.00074069225,0.0007843191,0.0010565001,0.0017166351],"category_scores_gemma":[0.019435653,0.00047851523,0.0004642044,0.00029685858,0.0009233728,0.0015394029,0.0010997296,0.0013100816,0.00028099483],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00077548914,0.00026482853,0.025274267,0.00023432633,0.00017931673,0.00037451615,0.00044645835,0.8380653,0.0076111774,0.0061553987,0.0009110942,0.119707786],"study_design_scores_gemma":[0.000025831196,0.00012846843,0.0069275107,0.000016585784,0.000013982195,0.000100312995,0.000045422057,0.9832758,0.0018465418,0.0074029826,0.00019377185,0.000022767119],"about_ca_topic_score_codex":0.0027192675,"about_ca_topic_score_gemma":0.0035642954,"teacher_disagreement_score":0.0027192675,"about_ca_system_score_codex":0.00056708563,"about_ca_system_score_gemma":0.0006180439,"threshold_uncertainty_score":0.013272107},"labels":[],"label_agreement":null},{"id":"W3103966376","doi":"","title":"Value-driven Hindsight Modelling","year":2020,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Hindsight bias; Leverage (statistics); Computer science; Reinforcement learning; Machine learning; Artificial intelligence; Bellman equation; Exploit; Function (biology); Representation (politics); Mathematical optimization; Mathematics","score_opus":0.03967133682234385,"score_gpt":0.2352821196128032,"score_spread":0.19561078279045935,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3103966376","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0153039405,0.00034520635,0.97197706,0.0012922062,0.00008908684,0.000071583694,0.00032719938,0.0003763613,0.010217282],"genre_scores_gemma":[0.8442179,0.0005997323,0.12952067,0.0005748778,0.00017794961,0.00032216765,0.00047421784,0.00024128138,0.023871299],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99842227,0.000640297,0.00006751217,0.00037927047,0.0002878365,0.00020286893],"domain_scores_gemma":[0.9941795,0.003802243,0.00070833106,0.00047773562,0.0005206157,0.00031150656],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027461606,0.001259297,0.0015249932,0.0006877886,0.0005083992,0.0023828163,0.002744855,0.002209631,0.008960513],"category_scores_gemma":[0.013841501,0.0006468284,0.0009899465,0.0008424637,0.0026344638,0.00302332,0.0022539278,0.0034611728,0.0013989885],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009755873,0.000046217807,0.00095432665,0.00010018555,0.00004701239,0.0001639564,0.00017176158,0.6772872,0.00075215945,0.2944474,0.0029441088,0.022988046],"study_design_scores_gemma":[0.00001471047,0.000020850168,0.00008989436,0.0000123527625,0.000005425834,0.000023122147,0.000010779389,0.8439957,0.00016055767,0.15466717,0.0009887003,0.000010815615],"about_ca_topic_score_codex":0.0050086114,"about_ca_topic_score_gemma":0.005762549,"teacher_disagreement_score":0.008960513,"about_ca_system_score_codex":0.0017938628,"about_ca_system_score_gemma":0.0015099857,"threshold_uncertainty_score":0.029975891},"labels":[],"label_agreement":null},{"id":"W3104132897","doi":"","title":"Deep Reinforcement and InfoMax Learning","year":2020,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Infomax; Reinforcement learning; Computer science; Artificial intelligence; Task (project management); Representation (politics); Machine learning; Mutual information; Baseline (sea); Channel (broadcasting)","score_opus":0.05192380332967931,"score_gpt":0.1673757088892988,"score_spread":0.11545190555961951,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3104132897","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08111986,0.00067822327,0.906459,0.0014913407,0.0001273871,0.00006433182,0.00032390998,0.0019604529,0.0077753784],"genre_scores_gemma":[0.8986264,0.00017631536,0.096915215,0.00033302247,0.00006448563,0.000103219376,0.00026327418,0.00009821348,0.0034198065],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99952245,0.00017121749,0.000022336286,0.00011472219,0.00009902903,0.00007029003],"domain_scores_gemma":[0.99830437,0.0009961065,0.00019011195,0.00020974592,0.00016252331,0.00013717063],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016442875,0.0010951034,0.0007948678,0.0004173072,0.0002868951,0.00079421257,0.0017605863,0.0013543874,0.0024383976],"category_scores_gemma":[0.0051420014,0.00035026457,0.00039408976,0.0003891944,0.0014812611,0.0016678022,0.0014138253,0.0022402168,0.00043954953],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026288594,0.000119135235,0.0011765494,0.000093769464,0.00006233337,0.00007032834,0.00004306595,0.90773106,0.002669583,0.030806094,0.0023585546,0.054606542],"study_design_scores_gemma":[0.000010832775,0.000030387439,0.00008388905,0.000005406404,0.000003921908,0.0000066325347,0.0000021573799,0.98936206,0.000906634,0.009297369,0.00028657774,0.000004190786],"about_ca_topic_score_codex":0.0027236417,"about_ca_topic_score_gemma":0.00329351,"teacher_disagreement_score":0.0027236417,"about_ca_system_score_codex":0.0012123246,"about_ca_system_score_gemma":0.0011586518,"threshold_uncertainty_score":0.008796096},"labels":[],"label_agreement":null},{"id":"W3105702366","doi":"","title":"Variational Policy Gradient Method for Reinforcement Learning with General Utilities","year":2020,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Mathematical optimization; Reinforcement learning; Convexity; Markov decision process; Q-learning; Mathematics; Gradient descent; Gradient method; Computer science; Applied mathematics; Convergence (economics); Markov process; Artificial neural network; Artificial intelligence; Finance","score_opus":0.03067472875020712,"score_gpt":0.28240915646595416,"score_spread":0.25173442771574706,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3105702366","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018731169,0.00024013974,0.99571866,0.00021537731,0.00003649751,0.000033896293,0.000019976378,0.0000860828,0.0017763159],"genre_scores_gemma":[0.41957343,0.0010707321,0.5589846,0.00048703418,0.00018170748,0.0007639742,0.0002646429,0.00048319684,0.018190762],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993724,0.00032411233,0.000021937096,0.000086876826,0.00013981877,0.000054873894],"domain_scores_gemma":[0.9984523,0.0011626858,0.00006995718,0.00006584479,0.00018409673,0.00006503834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022916833,0.0011166933,0.0013461533,0.00070612086,0.00041455834,0.0010316776,0.001399044,0.0014935981,0.0040505277],"category_scores_gemma":[0.00581927,0.0006667115,0.00080648775,0.00057924003,0.0016926489,0.001345704,0.0016839918,0.0024352078,0.0006902316],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004076752,0.000032062548,0.00029447046,0.00009556292,0.00003928955,0.00006238296,0.00006446524,0.77164894,0.00083357457,0.2065126,0.0013370975,0.019038826],"study_design_scores_gemma":[0.000005196491,0.0000056684344,0.000014896982,0.0000046704436,0.0000018320202,0.0000036573808,0.0000025816214,0.9804669,0.00008627094,0.018882219,0.0005237366,0.0000023845878],"about_ca_topic_score_codex":0.007117778,"about_ca_topic_score_gemma":0.0046133962,"teacher_disagreement_score":0.007117778,"about_ca_system_score_codex":0.0020983755,"about_ca_system_score_gemma":0.0023484677,"threshold_uncertainty_score":0.015224874},"labels":[],"label_agreement":null},{"id":"W3107153805","doi":"10.1038/s41586-020-2939-8","title":"Autonomous navigation of stratospheric balloons using reinforcement learning","year":2020,"lang":"en","type":"article","venue":"Nature","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":272,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Google (Canada)","funders":"","keywords":"Reinforcement learning; Computer science; Controller (irrigation); Obstacle avoidance; Process (computing); Obstacle; Reinforcement; Real-time computing; Artificial intelligence; Simulation; Engineering","score_opus":0.018149712530883974,"score_gpt":0.2613037861494292,"score_spread":0.24315407361854519,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3107153805","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18094963,0.0004026699,0.8118282,0.0004650708,0.000117671894,0.00004497914,0.000031141437,0.000548619,0.005611976],"genre_scores_gemma":[0.97694075,0.00006760935,0.021891732,0.000037013597,0.000013277492,0.000022989303,0.000016839589,0.000014547677,0.0009952608],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998683,0.000049104983,0.000005249321,0.00002498924,0.000029814797,0.000022601747],"domain_scores_gemma":[0.9994272,0.00030735528,0.00008560234,0.000043294032,0.00008747062,0.000049140323],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004528383,0.00032901278,0.0004206077,0.00021290075,0.00028649028,0.0004278577,0.00054558646,0.0005574573,0.00075488706],"category_scores_gemma":[0.0018640318,0.00020746904,0.00024053162,0.00015540574,0.0007902462,0.00046835645,0.0006220657,0.00069576007,0.00012435442],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006672545,0.00003282067,0.0010030313,0.000019134746,0.000022276938,0.00003159546,0.00003363698,0.96934545,0.002501015,0.0045005134,0.00037955563,0.022064222],"study_design_scores_gemma":[0.000007026286,0.000020476198,0.00009229728,0.000001713279,0.0000024520605,0.000004731431,0.0000032129374,0.99754363,0.00028011127,0.0018984538,0.00014374685,0.000002136719],"about_ca_topic_score_codex":0.005956338,"about_ca_topic_score_gemma":0.0042799558,"teacher_disagreement_score":0.005956338,"about_ca_system_score_codex":0.00046049771,"about_ca_system_score_gemma":0.0006066091,"threshold_uncertainty_score":0.011843383},"labels":[],"label_agreement":null},{"id":"W3107258301","doi":"10.48550/arxiv.2012.01418","title":"Team Optimal Control of Coupled Subsystems with Mean-Field Sharing","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Control theory (sociology); Optimal control; Dynamic programming; State (computer science); Mean field theory; Computer science; Controller (irrigation); Mathematical optimization; Field (mathematics); Stochastic control; Mathematics; Control (management); Algorithm; Artificial intelligence","score_opus":0.05192717160608362,"score_gpt":0.18326112596299296,"score_spread":0.13133395435690934,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3107258301","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23371929,0.00032645356,0.7569791,0.0007125646,0.00007774084,0.00006983386,0.00011603319,0.00018394107,0.007815031],"genre_scores_gemma":[0.98423624,0.000072168295,0.013174328,0.00006213694,0.000017088783,0.00007713519,0.000045039873,0.000020562627,0.0022953383],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9992267,0.0003090399,0.00002602043,0.00016296507,0.00011024466,0.00016508777],"domain_scores_gemma":[0.99719363,0.00187281,0.0003986218,0.00010276475,0.00021268691,0.000219543],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016604705,0.0011078122,0.0015832304,0.00046154228,0.00047597836,0.001008052,0.0009577089,0.0011322809,0.0016548708],"category_scores_gemma":[0.0038172056,0.00055097515,0.0008412664,0.0004032226,0.0017206954,0.0008942151,0.0019218654,0.0011043445,0.00015322072],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000070492155,0.000026545793,0.00031533308,0.000026555465,0.000042777923,0.00006322186,0.000034340526,0.9877758,0.0006541272,0.008991231,0.00016307455,0.0018364014],"study_design_scores_gemma":[0.000013326373,0.000028330454,0.00005980919,0.000002094975,0.0000048024795,0.000003734198,0.000008357257,0.9940826,0.00009991854,0.0056258375,0.00006783748,0.0000033122808],"about_ca_topic_score_codex":0.008610914,"about_ca_topic_score_gemma":0.0036248807,"teacher_disagreement_score":0.008610914,"about_ca_system_score_codex":0.0012391378,"about_ca_system_score_gemma":0.0010712551,"threshold_uncertainty_score":0.017121553},"labels":[],"label_agreement":null},{"id":"W3108136459","doi":"10.48550/arxiv.2011.12363","title":"C-Learning: Horizon-Aware Cumulative Accessibility Estimation","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reachability; Reinforcement learning; Computer science; Time horizon; Generalization; Motion planning; Sample (material); Reliability (semiconductor); Set (abstract data type); Path (computing); Horizon; Code (set theory); Monotonic function; Machine learning; Mathematical optimization; Artificial intelligence; Robot; Theoretical computer science; Mathematics","score_opus":0.09698605659490178,"score_gpt":0.23354127048946952,"score_spread":0.13655521389456773,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3108136459","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005943032,0.00015399909,0.9908976,0.00010582194,0.000027218433,0.000055959335,0.00008903532,0.0015494678,0.0011777802],"genre_scores_gemma":[0.5846868,0.00028330498,0.40916815,0.00030035488,0.00010157411,0.000405046,0.0006096557,0.0007192705,0.0037257145],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99900526,0.00022873975,0.00005212352,0.00027046702,0.00029604367,0.00014736765],"domain_scores_gemma":[0.99442506,0.003614022,0.00039397573,0.0005925141,0.00063588616,0.00033861087],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015861742,0.0012412091,0.001561463,0.0011938404,0.0006319107,0.0011035921,0.0027643077,0.0015397142,0.0057942527],"category_scores_gemma":[0.010044803,0.00061715214,0.0008248979,0.00090835034,0.0013200941,0.0017787336,0.002421022,0.002633998,0.0009060696],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001538693,0.00009407492,0.0010478714,0.00012762116,0.000042776806,0.00006782984,0.00006543381,0.8346398,0.0013564533,0.011531519,0.003637223,0.14723548],"study_design_scores_gemma":[0.000012534904,0.000016944512,0.000067733585,0.000007879185,0.0000040122577,0.000011339602,0.00000397678,0.99194604,0.00047819086,0.007072027,0.00037439232,0.000005050393],"about_ca_topic_score_codex":0.011386094,"about_ca_topic_score_gemma":0.012329508,"teacher_disagreement_score":0.011386094,"about_ca_system_score_codex":0.0014924805,"about_ca_system_score_gemma":0.003154555,"threshold_uncertainty_score":0.022639632},"labels":[],"label_agreement":null},{"id":"W3108183475","doi":"","title":"Skill Transfer via Partially Amortized Hierarchical Planning","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Suite; Reinforcement learning; Leverage (statistics); Task (project management); Adaptation (eye); Knowledge transfer; Amortization; Transfer of learning; Artificial intelligence; Machine learning; Human–computer interaction; Knowledge management","score_opus":0.08606900475964419,"score_gpt":0.20635410698759374,"score_spread":0.12028510222794955,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3108183475","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13187566,0.00042896566,0.8569911,0.0006746946,0.00008987682,0.0002834468,0.00020712252,0.0043421034,0.005107057],"genre_scores_gemma":[0.90380967,0.00009251801,0.09311488,0.00021832854,0.000036007605,0.00029146904,0.000252945,0.0001593408,0.0020249095],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989243,0.0003563012,0.00006365026,0.00027232905,0.00020467995,0.0001785865],"domain_scores_gemma":[0.9966605,0.0018528573,0.00026097722,0.0006851308,0.00027026588,0.00027027517],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025067541,0.0015939645,0.00139137,0.0007311124,0.0005947072,0.0010392249,0.002692449,0.0013844314,0.003388706],"category_scores_gemma":[0.0065479786,0.0008961696,0.00075602025,0.000596212,0.0015845327,0.0018620223,0.002636087,0.0019187332,0.00072748866],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002705615,0.0001857043,0.00094004744,0.00005931155,0.000059277896,0.000064210326,0.00007957809,0.92713654,0.002397202,0.0049307346,0.001559968,0.06231685],"study_design_scores_gemma":[0.000025366518,0.000046373418,0.000075672324,0.0000031433067,0.0000065271356,0.000007109377,0.000005203797,0.9939383,0.00040365694,0.0053736013,0.00011076154,0.0000042645866],"about_ca_topic_score_codex":0.0074042506,"about_ca_topic_score_gemma":0.005901418,"teacher_disagreement_score":0.0074042506,"about_ca_system_score_codex":0.0016835749,"about_ca_system_score_gemma":0.002074846,"threshold_uncertainty_score":0.014722288},"labels":[],"label_agreement":null},{"id":"W3110289401","doi":"10.1017/9781009302180.029","title":"Dynamic Programming Algorithms","year":2024,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Computer science; Dynamic programming; Algorithm","score_opus":0.017629680580646116,"score_gpt":0.2132394704960128,"score_spread":0.19560978991536668,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3110289401","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00084259507,0.011740499,0.7018005,0.0043899333,0.0015354166,0.00016565192,0.0007841131,0.0026032957,0.27613807],"genre_scores_gemma":[0.02654566,0.020903554,0.5896666,0.0038431676,0.0012467552,0.00091813796,0.0025069,0.002181604,0.35218757],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9992009,0.00014406351,0.000046060708,0.00016035499,0.00039885857,0.000049723294],"domain_scores_gemma":[0.99943167,0.00029928648,0.00002292877,0.00010006455,0.00011952813,0.000026607557],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00078976364,0.0013931268,0.0007218138,0.0007583296,0.00048498524,0.0030108357,0.0013930809,0.0012645255,0.05790508],"category_scores_gemma":[0.0030038573,0.0005580266,0.0007769476,0.0011262804,0.0010710281,0.003102511,0.0016560347,0.0039688847,0.040448934],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000014513898,0.000049866725,0.00014069653,0.0004929638,0.000027209357,0.000059550755,0.00017561359,0.0072944984,0.0009619133,0.40143248,0.23229106,0.35705957],"study_design_scores_gemma":[0.000011468836,0.00002167856,0.00015379312,0.0002937085,0.000007436457,0.00021818363,0.00004671832,0.012125665,0.00065460487,0.2775767,0.70887387,0.000016125168],"about_ca_topic_score_codex":0.00059347786,"about_ca_topic_score_gemma":0.0008501253,"teacher_disagreement_score":0.05790508,"about_ca_system_score_codex":0.0010048413,"about_ca_system_score_gemma":0.0009580864,"threshold_uncertainty_score":0.19371176},"labels":[],"label_agreement":null},{"id":"W3112448993","doi":"10.1109/smc42975.2020.9283018","title":"Motion Path Planning of Two Robot Arms in a Common Workspace","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Regina","funders":"","keywords":"Workspace; Motion planning; Computer science; Reinforcement learning; Discretization; Robot; Path (computing); Artificial intelligence; Context (archaeology); Kinematics; Collision; Q-learning; Mathematics; Physics","score_opus":0.03983504067494486,"score_gpt":0.2884268183005159,"score_spread":0.24859177762557105,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3112448993","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.205804,0.00010954951,0.79044527,0.00012638868,0.000021744352,0.00014865478,0.00003715682,0.0004390095,0.0028683434],"genre_scores_gemma":[0.8589643,0.00003957765,0.13961938,0.00002156064,0.0000032005767,0.0001548365,0.000055752254,0.0000240946,0.0011172171],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99962616,0.000111462534,0.000019119905,0.00009768552,0.0000795823,0.00006597894],"domain_scores_gemma":[0.9988657,0.0006076274,0.00015179344,0.00012863243,0.00013115873,0.000115168004],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00090702495,0.0005826895,0.00048607882,0.000363209,0.00043269913,0.00044085266,0.000870411,0.0008156699,0.0017987209],"category_scores_gemma":[0.0025205729,0.000362824,0.0004956616,0.00029757153,0.0008360676,0.0009486267,0.0010529571,0.00065252447,0.00020848733],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002029399,0.000118482,0.000886956,0.000053146276,0.000029380324,0.00018376244,0.00021599689,0.95194536,0.006896111,0.0065665618,0.00019879958,0.032702588],"study_design_scores_gemma":[0.000029262677,0.0001113976,0.00022265788,0.000005846728,0.0000046074097,0.000022511165,0.000032529857,0.9934668,0.0020664,0.0036696834,0.00036290198,0.0000054298594],"about_ca_topic_score_codex":0.003944099,"about_ca_topic_score_gemma":0.002393938,"teacher_disagreement_score":0.003944099,"about_ca_system_score_codex":0.000631998,"about_ca_system_score_gemma":0.0011726036,"threshold_uncertainty_score":0.007842302},"labels":[],"label_agreement":null},{"id":"W3117215073","doi":"10.1613/jair.1.13673","title":"Towards Continual Reinforcement Learning: A Review and Perspectives","year":2022,"lang":"en","type":"review","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":180,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; McGill University","funders":"Canada Excellence Research Chairs, Government of Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Computer science; Scope (computer science); Artificial intelligence; Bridging (networking); Taxonomy (biology); Function (biology); Management science; Engineering","score_opus":0.3537488428002021,"score_gpt":0.4758490705427069,"score_spread":0.12210022774250484,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3117215073","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00017915067,0.9949846,0.002204542,0.0006094696,0.00016892086,0.0000070957703,0.000010057894,0.000017184027,0.001819003],"genre_scores_gemma":[0.0026374736,0.99425995,0.001881602,0.00036712998,0.00034713798,0.000016084212,0.000025349571,0.000008224983,0.0004569902],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9994215,0.00013821546,0.00008101267,0.00014846,0.00017720074,0.000033736982],"domain_scores_gemma":[0.99679595,0.002390386,0.00016800096,0.000075192016,0.000472509,0.00009806215],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017961604,0.0011545583,0.0014095568,0.0026751643,0.00040747793,0.0021655406,0.0015585924,0.001961206,0.0038969803],"category_scores_gemma":[0.003938644,0.0006143135,0.0007157455,0.0044232197,0.0011327696,0.003142055,0.0010512057,0.002958983,0.0019911646],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000076871984,0.00014495711,0.0003916822,0.02293243,0.00012412561,0.0001457375,0.00023026246,0.0036418312,0.0005828555,0.0431855,0.015768971,0.91277486],"study_design_scores_gemma":[0.00003041949,0.0002574795,0.0011063331,0.015094548,0.00019056148,0.0009705718,0.00029368172,0.0025591739,0.0006364376,0.03512603,0.94364387,0.0000908437],"about_ca_topic_score_codex":0.0021053215,"about_ca_topic_score_gemma":0.0018657171,"teacher_disagreement_score":0.0038969803,"about_ca_system_score_codex":0.0013300007,"about_ca_system_score_gemma":0.002184543,"threshold_uncertainty_score":0.013036728},"labels":[],"label_agreement":null},{"id":"W3118340619","doi":"10.1109/ssci47803.2020.9308403","title":"Objective Comparison and Selection in Mono- and Multi-Objective Evolutionary Neurocontrollers","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Evolutionary algorithm; Mathematical optimization; Multi-objective optimization; Computer science; Pareto principle; Selection (genetic algorithm); Compatibility (geochemistry); Objectivity (philosophy); Mathematics; Artificial intelligence; Engineering","score_opus":0.023410978597037852,"score_gpt":0.2552927826432486,"score_spread":0.23188180404621075,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3118340619","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2470165,0.0006900114,0.7383612,0.00022470247,0.00008765796,0.00022588029,0.00003902312,0.00032728256,0.013027703],"genre_scores_gemma":[0.80695623,0.0001531043,0.18940887,0.00009342287,0.000014646009,0.00020457784,0.000043074528,0.000049000493,0.0030770358],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987338,0.0005015302,0.000069506015,0.00015087174,0.00046093238,0.00008336467],"domain_scores_gemma":[0.9978206,0.0013400988,0.0002437326,0.00015845984,0.00034141503,0.000095699346],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003868856,0.00088734354,0.00074168923,0.0010287502,0.00042554797,0.0011249658,0.0010112469,0.0007609965,0.00232632],"category_scores_gemma":[0.0059832307,0.00033559705,0.00045246075,0.00047481718,0.0008633314,0.00096864806,0.0010830386,0.00083605084,0.00019102305],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002475222,0.00021473726,0.002518013,0.00016490405,0.00014608045,0.00011472898,0.000108588436,0.84250474,0.009546646,0.020282619,0.00031868648,0.123832844],"study_design_scores_gemma":[0.000036962552,0.0002971842,0.0011253101,0.000024802635,0.000028008311,0.000054350705,0.000029447425,0.98845357,0.00376627,0.005259782,0.0009060391,0.000018351491],"about_ca_topic_score_codex":0.0008214469,"about_ca_topic_score_gemma":0.0012551879,"teacher_disagreement_score":0.003868856,"about_ca_system_score_codex":0.00093099574,"about_ca_system_score_gemma":0.00063275744,"threshold_uncertainty_score":0.020460725},"labels":[],"label_agreement":null},{"id":"W3118394271","doi":"10.1609/aaai.v35i11.17166","title":"Solving Common-Payoff Games with Approximate Policy Iteration","year":2021,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Stochastic game; Scalability; Artificial intelligence; Common knowledge (logic); Scale (ratio); Mathematical optimization; Mathematical economics; Mathematics","score_opus":0.04921359465533086,"score_gpt":0.28882868471805523,"score_spread":0.23961509006272436,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3118394271","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026910858,0.00018857357,0.966407,0.0004552162,0.000040739174,0.000083847175,0.000039864204,0.00045853347,0.005415401],"genre_scores_gemma":[0.7468053,0.00017551906,0.24793416,0.000305024,0.000050534738,0.00034693268,0.00013899208,0.0001250662,0.0041183326],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987458,0.00050285767,0.00006186396,0.00024553927,0.00025770508,0.00018609816],"domain_scores_gemma":[0.9946477,0.004128562,0.00031404095,0.00036882827,0.00029679967,0.00024407277],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023133766,0.0011976748,0.0016489358,0.00062044093,0.0006138362,0.0013216346,0.0015947022,0.0016087889,0.002347057],"category_scores_gemma":[0.010751815,0.0007520452,0.0006743508,0.00054473145,0.0020582215,0.0019340779,0.0023110667,0.002634974,0.0004301044],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007579671,0.00006756953,0.00071549363,0.000060566912,0.000043404805,0.000046001667,0.00009594943,0.918689,0.0003528968,0.052788556,0.0009772474,0.026087502],"study_design_scores_gemma":[0.000013794904,0.0000144726755,0.00002649962,0.000004860173,0.0000031681989,0.000005985622,0.000008366132,0.97595745,0.00013948913,0.02357583,0.00024688835,0.0000032892915],"about_ca_topic_score_codex":0.0060786745,"about_ca_topic_score_gemma":0.007838199,"teacher_disagreement_score":0.0060786745,"about_ca_system_score_codex":0.0018123913,"about_ca_system_score_gemma":0.002990687,"threshold_uncertainty_score":0.013149917},"labels":[],"label_agreement":null},{"id":"W3119430014","doi":"10.65109/bwtd6573","title":"Partially Observable Mean Field Reinforcement Learning","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Scalability; Field (mathematics); Visibility; Artificial intelligence; Observable; Mathematical optimization; Set (abstract data type); Q-learning; Machine learning; Mathematics","score_opus":0.03734177779240893,"score_gpt":0.2667191034409659,"score_spread":0.22937732564855695,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3119430014","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03858088,0.0002390627,0.9572496,0.00047007747,0.000069932816,0.00006620095,0.00006457226,0.00057376,0.0026860079],"genre_scores_gemma":[0.9287167,0.000112221984,0.06788044,0.00018889834,0.000047454014,0.0001351826,0.00009553202,0.000045403744,0.0027780957],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988493,0.00047837698,0.000042945463,0.00024243543,0.00022772182,0.00015930053],"domain_scores_gemma":[0.99317694,0.004833802,0.0006220611,0.0004447878,0.00058402447,0.00033835703],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002311522,0.0011067068,0.0016052871,0.00048369204,0.000506138,0.00096432015,0.0019991659,0.0015166606,0.001902437],"category_scores_gemma":[0.00960112,0.0004747375,0.0005252878,0.0005041896,0.0018369447,0.0014176692,0.0012330231,0.0019411959,0.00032929803],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000787707,0.000045929984,0.0005341673,0.000036883437,0.000026268084,0.000044137083,0.00003003138,0.97059995,0.00036891442,0.014439611,0.000690599,0.013104766],"study_design_scores_gemma":[0.0000113629285,0.000012771604,0.00003579822,0.0000020347484,0.0000021387932,0.000003640445,0.0000020238585,0.9936748,0.000095746116,0.0060482803,0.000108945904,0.0000024160402],"about_ca_topic_score_codex":0.0051551806,"about_ca_topic_score_gemma":0.004260678,"teacher_disagreement_score":0.0051551806,"about_ca_system_score_codex":0.0014944518,"about_ca_system_score_gemma":0.0014961141,"threshold_uncertainty_score":0.012224674},"labels":[],"label_agreement":null},{"id":"W3119528129","doi":"10.1109/ssci47803.2020.9308450","title":"Self-Adaptation of Meta-Parameters for Lamarckian-Inherited Neuromodulated Neurocontrollers in the Pursuit-Evasion Game","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Evolutionary algorithm; Adaptation (eye); Computer science; Evolutionary computation; Set (abstract data type); Pursuit-evasion; Evolution strategy; Selection (genetic algorithm); Artificial intelligence; Mathematical optimization; Mathematics","score_opus":0.10416389623534689,"score_gpt":0.26003691895567027,"score_spread":0.15587302272032338,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3119528129","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.62277883,0.0005066955,0.36608952,0.00021567079,0.0000752821,0.000147063,0.00004930787,0.00090956845,0.00922798],"genre_scores_gemma":[0.9641535,0.00005041572,0.034559205,0.00003942842,0.0000027839137,0.00008578866,0.000021145106,0.000028293685,0.0010595071],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99983835,0.000036228317,0.000011308175,0.00003271027,0.000049687496,0.000031599622],"domain_scores_gemma":[0.9995783,0.00014204431,0.000090993526,0.000060979684,0.00009843148,0.000029259854],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00052840787,0.00058736076,0.00034280322,0.00038138172,0.00027859057,0.00057127106,0.0010903614,0.00065086986,0.001003069],"category_scores_gemma":[0.0015113299,0.00022847399,0.0003367388,0.00013816539,0.0004715325,0.00042752933,0.0005925659,0.00071106973,0.00019449614],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000981622,0.00014417573,0.0024854147,0.00008129803,0.000087128436,0.00024827212,0.0001557629,0.90293217,0.048293132,0.00388399,0.0003145117,0.041275933],"study_design_scores_gemma":[0.000021741333,0.00014840439,0.0008102455,0.000016907383,0.00003285759,0.00009571069,0.000033054937,0.981263,0.015562336,0.0010745337,0.00092190335,0.00001930784],"about_ca_topic_score_codex":0.0014362693,"about_ca_topic_score_gemma":0.0020010455,"teacher_disagreement_score":0.0014362693,"about_ca_system_score_codex":0.0006906616,"about_ca_system_score_gemma":0.00050371065,"threshold_uncertainty_score":0.005011201},"labels":[],"label_agreement":null},{"id":"W3120394327","doi":"10.1007/s10515-021-00313-x","title":"Faults in deep reinforcement learning programs: a taxonomy and a detection approach","year":2021,"lang":"en","type":"article","venue":"Automated Software Engineering","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Fonds de Recherche du Québec-Société et Culture; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Computer science; Categorization; Artificial intelligence; Fault detection and isolation; Software; Machine learning; Software engineering; Programming language","score_opus":0.012062613245359722,"score_gpt":0.20269778829536153,"score_spread":0.1906351750500018,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3120394327","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03495555,0.0019769568,0.9594291,0.0007904185,0.00004332014,0.00008715279,0.0000961708,0.00074370374,0.0018775761],"genre_scores_gemma":[0.78872323,0.0017576963,0.206377,0.00024034093,0.00019136611,0.0001690634,0.00018438493,0.0001241721,0.0022328012],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9966654,0.00081146584,0.00040179057,0.00061941124,0.0012278702,0.00027389414],"domain_scores_gemma":[0.9746957,0.018112179,0.0030275914,0.0017756018,0.0019553958,0.0004335797],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029227575,0.0009819542,0.0010527857,0.0024418323,0.00051834004,0.0023785208,0.0024866043,0.0034986692,0.0009522244],"category_scores_gemma":[0.019584794,0.0008892048,0.000962648,0.0019371427,0.003147982,0.004881529,0.001945743,0.0028991525,0.00016889173],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041421817,0.00033301575,0.020295475,0.0008640878,0.00012904858,0.0009581085,0.000653554,0.3742052,0.00456732,0.324276,0.0027468326,0.2705571],"study_design_scores_gemma":[0.000012719144,0.00008503461,0.0007017835,0.000067936984,0.000029656043,0.00028825935,0.00005012322,0.8528542,0.0022456837,0.14266415,0.0009804758,0.0000199686],"about_ca_topic_score_codex":0.0019706131,"about_ca_topic_score_gemma":0.0012867607,"teacher_disagreement_score":0.0034986692,"about_ca_system_score_codex":0.0015083789,"about_ca_system_score_gemma":0.0011131284,"threshold_uncertainty_score":0.015457213},"labels":[],"label_agreement":null},{"id":"W3123244128","doi":"10.1109/lars/sbr/wre51543.2020.9306954","title":"Deep Reinforcement Learning Applied to IEEE Very Small Size Soccer Strategy","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Conselho Nacional de Desenvolvimento Científico e Tecnológico","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Curriculum; Robot; Heuristic; Context (archaeology); Adversary; Machine learning","score_opus":0.033649539412251794,"score_gpt":0.24077473726701018,"score_spread":0.20712519785475839,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3123244128","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17251107,0.0003992829,0.8177993,0.00034408475,0.00012927639,0.00012234625,0.00004821639,0.0009782469,0.007668118],"genre_scores_gemma":[0.96967286,0.00005066846,0.028397765,0.00005339966,0.00000869119,0.00004959759,0.000026381733,0.00001762735,0.0017230817],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997873,0.00006692541,0.0000109131715,0.0000383348,0.000048793572,0.000047714526],"domain_scores_gemma":[0.9995177,0.00021651,0.000056178105,0.000035409368,0.00012426118,0.000050059512],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00072839006,0.0005065097,0.00045192929,0.00024057449,0.00020913499,0.00039438176,0.0006851233,0.00051659474,0.0015058181],"category_scores_gemma":[0.0018091705,0.00021861901,0.00022922984,0.00012943827,0.0005368917,0.0003304336,0.0006273094,0.0007966684,0.00018155065],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000049307528,0.000067399706,0.00064584485,0.00003610131,0.000019496865,0.000043795466,0.000031368447,0.9588974,0.0023152195,0.0027776621,0.00035466463,0.034761705],"study_design_scores_gemma":[0.000003852175,0.00003190952,0.0000837459,0.0000021348403,0.0000021045423,0.00000381175,0.000002322553,0.99876547,0.00039356915,0.00057727686,0.00013224134,0.0000016106445],"about_ca_topic_score_codex":0.0075227153,"about_ca_topic_score_gemma":0.00589927,"teacher_disagreement_score":0.0075227153,"about_ca_system_score_codex":0.0007166212,"about_ca_system_score_gemma":0.0010737915,"threshold_uncertainty_score":0.014957845},"labels":[],"label_agreement":null},{"id":"W3126150352","doi":"10.48550/arxiv.2102.03718","title":"An Analysis of Frame-skipping in Reinforcement Learning","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Frame (networking); Task (project management); Inertia; Offset (computer science); Consistency (knowledge bases); Action (physics); Artificial intelligence; Machine learning; Algorithm","score_opus":0.05482070555975511,"score_gpt":0.20925271551283764,"score_spread":0.15443200995308254,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3126150352","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039426282,0.0012700957,0.9484927,0.000716445,0.000087732755,0.00012263804,0.00009195086,0.00038103957,0.009411172],"genre_scores_gemma":[0.8896659,0.0010287742,0.10020168,0.00044067128,0.000203834,0.00043123952,0.00018799213,0.00025225795,0.0075876703],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980444,0.0007073873,0.00008330927,0.00033619034,0.00054267404,0.00028611015],"domain_scores_gemma":[0.9888334,0.008574435,0.0008236003,0.00059748295,0.0006695102,0.0005014988],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005292586,0.0014263177,0.0014167029,0.00079836283,0.0007457297,0.0011290174,0.0023138842,0.0016441983,0.006024861],"category_scores_gemma":[0.021181189,0.0007041519,0.00094885495,0.0006459443,0.0020554808,0.0024689832,0.0016531983,0.0030392853,0.00042849124],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026821654,0.00014264294,0.001372931,0.0002228881,0.000067153815,0.00018522375,0.00019153507,0.7989452,0.0024520978,0.15278476,0.0019069848,0.041460387],"study_design_scores_gemma":[0.000015494286,0.00007002753,0.00017400099,0.000016036287,0.000011676064,0.00001817921,0.000006068205,0.9741072,0.00026215595,0.024897853,0.0004133394,0.000007931538],"about_ca_topic_score_codex":0.006957761,"about_ca_topic_score_gemma":0.0037634145,"teacher_disagreement_score":0.006957761,"about_ca_system_score_codex":0.0026183284,"about_ca_system_score_gemma":0.0017855222,"threshold_uncertainty_score":0.027990162},"labels":[],"label_agreement":null},{"id":"W3126457215","doi":"10.1007/s00521-023-09096-6","title":"hammer: Multi-level coordination of reinforcement learning agents via learned messaging","year":2023,"lang":"en","type":"article","venue":"Neural Computing and Applications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Representation (politics); Artificial intelligence; Information sharing; Distributed computing; Artificial neural network; Embodied agent; Human–computer interaction; Machine learning; Embodied cognition; World Wide Web","score_opus":0.07241711314584617,"score_gpt":0.32680994064118757,"score_spread":0.2543928274953414,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3126457215","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018455831,0.00011983977,0.96895427,0.00025690685,0.00017708585,0.000121797464,0.00007282694,0.0060367095,0.005804879],"genre_scores_gemma":[0.788222,0.00009051973,0.1995144,0.00017617093,0.00006355382,0.00027613042,0.00014082699,0.0003931642,0.011123177],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99931765,0.00018654812,0.000046516117,0.00016570222,0.00017700145,0.00010666692],"domain_scores_gemma":[0.99832934,0.00084054156,0.00012634405,0.00032340465,0.00017518982,0.00020515703],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012281892,0.0007706011,0.00097408047,0.00049445074,0.00064702047,0.001272656,0.002196142,0.0014958306,0.009532152],"category_scores_gemma":[0.0045315777,0.00052926457,0.00042057162,0.00035037662,0.0009782531,0.0016792614,0.002727699,0.0019127723,0.0014851233],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008840659,0.0003032499,0.0008995222,0.00020469182,0.00011739876,0.00039222836,0.00026174192,0.7424741,0.017499603,0.06706534,0.0089030545,0.16099505],"study_design_scores_gemma":[0.000059387505,0.000053433847,0.00005575653,0.000006318383,0.000007856309,0.000019203191,0.000013684498,0.98098433,0.002203637,0.0153225735,0.0012653746,0.000008452737],"about_ca_topic_score_codex":0.0020920846,"about_ca_topic_score_gemma":0.0027180612,"teacher_disagreement_score":0.009532152,"about_ca_system_score_codex":0.00064213044,"about_ca_system_score_gemma":0.00093763747,"threshold_uncertainty_score":0.031888247},"labels":[],"label_agreement":null},{"id":"W3127407414","doi":"","title":"Planning from Pixels using Inverse Dynamics Models","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Task (project management); Heuristic; Computer science; Focus (optics); Artificial intelligence; Machine learning; Dynamics (music); Engineering; Psychology","score_opus":0.19904125336702247,"score_gpt":0.21294462833526798,"score_spread":0.013903374968245508,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3127407414","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019388633,0.000117768184,0.9772997,0.00022343615,0.000027444914,0.000033023596,0.00013532651,0.0010815331,0.0016931542],"genre_scores_gemma":[0.7331048,0.00022519148,0.26094896,0.00015214928,0.000043354645,0.00017186649,0.0005355185,0.0003056316,0.0045124693],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996718,0.00007532392,0.000014240368,0.00013355109,0.000067913155,0.000037163758],"domain_scores_gemma":[0.9992142,0.0004604145,0.0000919105,0.0001043837,0.00006237186,0.00006657457],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00048708788,0.00093924825,0.00086582464,0.0005099295,0.00040805596,0.0011677219,0.001241782,0.0010471193,0.0037523715],"category_scores_gemma":[0.002402244,0.00083189836,0.00083927554,0.0004845989,0.0011020257,0.0016613156,0.0016263027,0.0017946032,0.0006271402],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000084416606,0.000027780941,0.0003967842,0.000037893362,0.000019822151,0.000058200676,0.000047537447,0.9575277,0.001447167,0.009079147,0.0010179344,0.030255591],"study_design_scores_gemma":[0.0000064874125,0.0000083939185,0.00003411717,0.0000030594144,0.0000020466862,0.0000061675746,0.0000053008034,0.98983777,0.00030825604,0.009569812,0.00021587734,0.0000027531144],"about_ca_topic_score_codex":0.009079265,"about_ca_topic_score_gemma":0.01392694,"teacher_disagreement_score":0.009079265,"about_ca_system_score_codex":0.0009218771,"about_ca_system_score_gemma":0.001363415,"threshold_uncertainty_score":0.018052876},"labels":[],"label_agreement":null},{"id":"W3127605640","doi":"10.1007/s00521-021-06375-y","title":"Improving reinforcement learning with human assistance: an argument for human subject studies with HIPPO Gym","year":2021,"lang":"en","type":"preprint","venue":"Neural Computing and Applications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; University of Alberta; Alberta Machine Intelligence Institute; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada; Amazon Web Services; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Human–computer interaction; Subject (documents); Parsing; World Wide Web","score_opus":0.04079369190192602,"score_gpt":0.3265417845465773,"score_spread":0.2857480926446513,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3127605640","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15280263,0.0020082458,0.76179516,0.024772597,0.0005549102,0.0004011601,0.0002962472,0.0017625745,0.05560652],"genre_scores_gemma":[0.9191466,0.00058819895,0.07006444,0.0024669133,0.00027849388,0.0003272815,0.00010432192,0.00019254757,0.0068311747],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9959318,0.0026321285,0.000084059626,0.00055232766,0.0006495832,0.0001501181],"domain_scores_gemma":[0.98200375,0.0125924,0.0007696001,0.0030219483,0.0009634746,0.000648867],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009300175,0.00080652116,0.0012676658,0.00043184657,0.00074964465,0.001109371,0.0015436375,0.0024888169,0.008239507],"category_scores_gemma":[0.03540395,0.00028394186,0.0004933809,0.00040169887,0.003711985,0.002934225,0.0024583172,0.0025542888,0.0009995274],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004547535,0.0016998035,0.017488169,0.0012750358,0.00081707723,0.00030343438,0.0017103651,0.10646623,0.015594341,0.2678783,0.021400735,0.560819],"study_design_scores_gemma":[0.00090467004,0.0025273098,0.015752496,0.00022709277,0.00023528423,0.00040766972,0.00053639757,0.29409325,0.013947347,0.6356454,0.03562536,0.00009773493],"about_ca_topic_score_codex":0.0019302865,"about_ca_topic_score_gemma":0.0013137715,"teacher_disagreement_score":0.009300175,"about_ca_system_score_codex":0.0007305677,"about_ca_system_score_gemma":0.0015603497,"threshold_uncertainty_score":0.04918462},"labels":[],"label_agreement":null},{"id":"W3127742036","doi":"10.48550/arxiv.2103.00336","title":"Transformers with Competitive Ensembles of Independent Mechanisms","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Transformer; Computer science; Engineering; Electrical engineering; Voltage","score_opus":0.041398407696539664,"score_gpt":0.17015250408307234,"score_spread":0.12875409638653268,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3127742036","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07435852,0.00042959364,0.91498846,0.00050544983,0.000092205206,0.00009979006,0.00008663877,0.0013241402,0.008115188],"genre_scores_gemma":[0.933211,0.00027462383,0.05870293,0.00023369273,0.00005563912,0.00012806503,0.00009736153,0.00014756396,0.007149061],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99912375,0.00022455788,0.000064473454,0.0002224526,0.00021580987,0.00014902011],"domain_scores_gemma":[0.99729997,0.0012405177,0.00027486324,0.0005889472,0.0003104558,0.00028533157],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021525729,0.0011669869,0.0011597002,0.00064243074,0.0004227585,0.001653862,0.0024225092,0.0014684197,0.0050793863],"category_scores_gemma":[0.006328718,0.0008605386,0.0015250506,0.00047905743,0.0021545917,0.004617186,0.0030222351,0.0025128948,0.0010200158],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058608217,0.00019500361,0.001999316,0.00025813538,0.00027986348,0.00031741947,0.00029121144,0.6102629,0.021609537,0.25022888,0.002851912,0.11111981],"study_design_scores_gemma":[0.0000364114,0.000089765344,0.00013010009,0.000009384607,0.000036628342,0.000052597836,0.000015922033,0.9366392,0.0023423359,0.059967656,0.00066316983,0.000016852837],"about_ca_topic_score_codex":0.0018057346,"about_ca_topic_score_gemma":0.0021420259,"teacher_disagreement_score":0.0050793863,"about_ca_system_score_codex":0.0012281138,"about_ca_system_score_gemma":0.0012398816,"threshold_uncertainty_score":0.016992271},"labels":[],"label_agreement":null},{"id":"W3128954025","doi":"10.1609/aaai.v35i9.16964","title":"Variance Penalized On-Policy and Off-Policy Actor-Critic","year":2021,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; McGill University","funders":"","keywords":"Variance (accounting); Convergence (economics); Computer science; Reinforcement learning; Estimator; Moment (physics); Markov decision process; Expected return; Mathematical optimization; Minimum-variance unbiased estimator; Variance risk premium; Econometrics; Mathematics; Artificial intelligence; Markov process; Statistics; Economics","score_opus":0.06320478828266483,"score_gpt":0.3209269653549211,"score_spread":0.25772217707225625,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3128954025","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017822072,0.00031753452,0.9762512,0.00026758396,0.000088723515,0.000064603955,0.00002805924,0.00067865074,0.004481612],"genre_scores_gemma":[0.7630672,0.00026860778,0.22678596,0.00030326465,0.00009311245,0.00019336081,0.00012977447,0.00034897643,0.0088097425],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99868923,0.00053932704,0.000056997218,0.00023680249,0.00034030588,0.00013732068],"domain_scores_gemma":[0.99647814,0.002016432,0.00042161412,0.00041314427,0.0004894939,0.0001812038],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025409698,0.0013201771,0.0014454776,0.00061314163,0.00040923222,0.0012523565,0.0019972678,0.0016924394,0.00237633],"category_scores_gemma":[0.008129375,0.00062507467,0.00054489763,0.0004913334,0.00136537,0.0011068976,0.0014058623,0.0021095835,0.0005695066],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011251285,0.00008143034,0.00049642334,0.000043492557,0.00004199445,0.000043114644,0.000037265887,0.9329308,0.0010647958,0.012173716,0.0012476629,0.0517268],"study_design_scores_gemma":[0.00000779591,0.00002194675,0.00004871882,0.000004365188,0.000003565268,0.000009727238,0.000002198583,0.9970605,0.00036664796,0.0021832709,0.00028760513,0.0000036794625],"about_ca_topic_score_codex":0.0029983004,"about_ca_topic_score_gemma":0.003178384,"teacher_disagreement_score":0.0029983004,"about_ca_system_score_codex":0.0012956836,"about_ca_system_score_gemma":0.0019661384,"threshold_uncertainty_score":0.013438106},"labels":[],"label_agreement":null},{"id":"W3130235457","doi":"10.1109/candarw51189.2020.00031","title":"FPGA Acceleration of ROS2-Based Reinforcement Learning Agents","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Field-programmable gate array; Robot; Artificial neural network; Robotics; Artificial intelligence; Embedded system; Latency (audio); Robot learning; Mobile robot","score_opus":0.05309554792447842,"score_gpt":0.2673577872631097,"score_spread":0.2142622393386313,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3130235457","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.45051974,0.00075537065,0.46331817,0.00032691882,0.00042976355,0.00036999458,0.0005667996,0.030063283,0.053649977],"genre_scores_gemma":[0.9491567,0.000068312314,0.043253947,0.000053229654,0.000012646624,0.00008603239,0.00018547835,0.00009770061,0.0070859455],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998561,0.000019255816,0.0000069125986,0.000030411344,0.00005971403,0.000027589283],"domain_scores_gemma":[0.99985075,0.000029365121,0.000018480987,0.000026029284,0.000056020137,0.000019397263],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00014921775,0.0005072757,0.00023803984,0.00023737621,0.00015705646,0.00034608514,0.000738443,0.00019622623,0.007461018],"category_scores_gemma":[0.00041134295,0.00015179272,0.000120283425,0.00011450073,0.00011854152,0.00021867406,0.00025741916,0.00029722703,0.0012110544],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022434343,0.0004091739,0.0065074293,0.0005851237,0.00012364403,0.0009913654,0.000282537,0.34247428,0.15498383,0.008238263,0.019340927,0.46382],"study_design_scores_gemma":[0.00019809678,0.0006820997,0.003761094,0.000033789962,0.00003447643,0.0002150513,0.000038205184,0.9011492,0.072253555,0.0008484389,0.020753551,0.00003254005],"about_ca_topic_score_codex":0.0037218798,"about_ca_topic_score_gemma":0.004011402,"teacher_disagreement_score":0.007461018,"about_ca_system_score_codex":0.00030827135,"about_ca_system_score_gemma":0.00040839685,"threshold_uncertainty_score":0.024959564},"labels":[],"label_agreement":null},{"id":"W3130843035","doi":"","title":"Conservative Safety Critics for Exploration","year":2021,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Convergence (economics); Task (project management); Suite; Upper and lower bounds; Artificial intelligence; Machine learning; Mathematical optimization; Engineering; Mathematics; Law","score_opus":0.11749702201540502,"score_gpt":0.379141317792146,"score_spread":0.261644295776741,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3130843035","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012740063,0.00018156301,0.98320353,0.00037872855,0.000045867062,0.000044180648,0.00004437142,0.00054183585,0.0028198424],"genre_scores_gemma":[0.8612038,0.0002989043,0.13127355,0.00048542809,0.00011281724,0.00038659055,0.00018916049,0.00027794216,0.005771711],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981446,0.0006519448,0.000089491434,0.00033468183,0.00058926846,0.00019006703],"domain_scores_gemma":[0.9866975,0.009744322,0.0010062671,0.0011624573,0.0009308944,0.00045856615],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034274259,0.0015237812,0.0012648767,0.0008666893,0.0007102448,0.0014982783,0.0018138998,0.0016763764,0.0032601508],"category_scores_gemma":[0.021400947,0.00075005664,0.0007847692,0.00035627605,0.0031859162,0.002227191,0.0031328248,0.0044276672,0.0006799951],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014473865,0.000043720895,0.0010805686,0.00011413781,0.00003647535,0.00012213849,0.00014876656,0.88915306,0.0020653128,0.08092644,0.0015884724,0.024576066],"study_design_scores_gemma":[0.000019769432,0.00004340584,0.00007399048,0.000022379494,0.000007249618,0.00003090813,0.000010621013,0.9561567,0.00071065786,0.042286925,0.0006281428,0.000009310877],"about_ca_topic_score_codex":0.0015100676,"about_ca_topic_score_gemma":0.001651699,"teacher_disagreement_score":0.0034274259,"about_ca_system_score_codex":0.0012905974,"about_ca_system_score_gemma":0.0020003982,"threshold_uncertainty_score":0.01812619},"labels":[],"label_agreement":null},{"id":"W3131011999","doi":"","title":"Memory-based Deep Reinforcement Learning for POMDP.","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Partially observable Markov decision process; Reinforcement learning; Computer science; Markov decision process; Observable; Artificial intelligence; Component (thermodynamics); Noise (video); Feature engineering; Process (computing); Robotics; Feature (linguistics); Machine learning; Deep learning; Markov process; Markov chain; Markov model; Robot; Mathematics","score_opus":0.06249373525342344,"score_gpt":0.19857316350256013,"score_spread":0.13607942824913669,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3131011999","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016310202,0.0005681406,0.9792322,0.00027689277,0.00007654698,0.00005808473,0.00011573957,0.0009474759,0.0024148263],"genre_scores_gemma":[0.8667476,0.00028812556,0.12981953,0.0002272539,0.000034833036,0.00021211663,0.0002527591,0.000097113516,0.0023206528],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996815,0.00008853037,0.000022078853,0.00007597838,0.00007213933,0.00005988271],"domain_scores_gemma":[0.9985826,0.00096487557,0.00013185362,0.00008975905,0.00015105188,0.00007982056],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009706444,0.0009859692,0.0010400509,0.0003768428,0.00033180395,0.0005718733,0.0012563978,0.000897947,0.0029260695],"category_scores_gemma":[0.003432118,0.0004798787,0.00054672104,0.00037450722,0.0007889193,0.0009332053,0.0012066684,0.0019873693,0.0003928727],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006422248,0.000058030622,0.0006251471,0.000095822616,0.00003489961,0.00004468136,0.000037485548,0.9450417,0.0006170596,0.00746083,0.001065669,0.04485448],"study_design_scores_gemma":[0.0000049375085,0.000011469469,0.00002260663,0.0000031798518,0.000002356687,0.0000035390415,0.000001905928,0.9968353,0.00014567284,0.0028345848,0.0001329819,0.0000014532012],"about_ca_topic_score_codex":0.008988316,"about_ca_topic_score_gemma":0.010015556,"teacher_disagreement_score":0.008988316,"about_ca_system_score_codex":0.0013780646,"about_ca_system_score_gemma":0.0016654434,"threshold_uncertainty_score":0.017872036},"labels":[],"label_agreement":null},{"id":"W3131546278","doi":"10.1609/aaai.v35i13.17378","title":"How RL Agents Behave When Their Actions Are Modified","year":2021,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Action (physics); Supervisor; Markov decision process; Computer science; Process (computing); Q-learning; Reinforcement; Control (management); Risk analysis (engineering); Intervention (counseling); Artificial intelligence; Markov process; Psychology; Business; Social psychology; Mathematics; Political science; Law","score_opus":0.21135287046727055,"score_gpt":0.31668126810195774,"score_spread":0.10532839763468718,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3131546278","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3597941,0.00053444266,0.61315346,0.0038699303,0.00017328406,0.00011328937,0.00016372178,0.0014413317,0.020756375],"genre_scores_gemma":[0.97106665,0.00019992219,0.025625268,0.00020314081,0.000017995622,0.00004727331,0.00005068659,0.00011503723,0.0026741105],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99881375,0.0005999251,0.00004860388,0.00019616712,0.00019908992,0.00014255207],"domain_scores_gemma":[0.9972362,0.0013430384,0.00041384532,0.00051399716,0.00031778408,0.00017518485],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017577402,0.00036203896,0.00035679864,0.00025860092,0.000302238,0.0014443637,0.0007712796,0.0011768528,0.0014348645],"category_scores_gemma":[0.011139401,0.00032945754,0.00030330982,0.00015760734,0.00148596,0.002329903,0.0006589775,0.0010548403,0.0005703611],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020054048,0.00009209545,0.009564803,0.000121170146,0.00014380009,0.00031894867,0.0006763741,0.8431384,0.0110527715,0.085722886,0.0024659906,0.046502277],"study_design_scores_gemma":[0.00003405642,0.00004473866,0.0013251912,0.000024529545,0.000021402559,0.00007143507,0.00012004104,0.9293588,0.002218972,0.06513131,0.0016214354,0.000028141183],"about_ca_topic_score_codex":0.004133081,"about_ca_topic_score_gemma":0.0024473208,"teacher_disagreement_score":0.004133081,"about_ca_system_score_codex":0.0008522718,"about_ca_system_score_gemma":0.00079485454,"threshold_uncertainty_score":0.00929594},"labels":[],"label_agreement":null},{"id":"W3132054471","doi":"10.48550/arxiv.2102.08607","title":"On the Convergence and Sample Efficiency of Variance-Reduced Policy Gradient Method","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Convexity; Truncation (statistics); Variance (accounting); Convergence (economics); Term (time); Mathematics; Variance reduction; Applied mathematics; Reinforcement learning; Mathematical optimization; Function (biology); Sample (material); Computer science; Statistics; Economics; Physics; Artificial intelligence; Monte Carlo method","score_opus":0.06947985645106997,"score_gpt":0.22228006022886093,"score_spread":0.15280020377779097,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3132054471","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016572904,0.00062569167,0.9779377,0.00057369843,0.00007144925,0.00010746362,0.000043565225,0.0004784839,0.0035890536],"genre_scores_gemma":[0.5908941,0.00090522034,0.40047565,0.00066809496,0.0001666117,0.00070063106,0.00032945228,0.00060081744,0.0052595115],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99713445,0.0015345545,0.000112111986,0.00030359832,0.000679885,0.00023541662],"domain_scores_gemma":[0.97825634,0.01802459,0.0006071403,0.0011392597,0.0015787185,0.00039393597],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0070958943,0.0012622084,0.0019946585,0.0010873212,0.00066323136,0.0012341981,0.0020169895,0.0016624723,0.003471499],"category_scores_gemma":[0.04499343,0.00069331063,0.0009504411,0.00058740244,0.002468071,0.0021828099,0.0025102657,0.0035769274,0.00077341986],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004299908,0.00021272372,0.0021980808,0.00024922725,0.00011133543,0.00013597708,0.0001599115,0.8390395,0.0026865022,0.093712114,0.0026127824,0.05845176],"study_design_scores_gemma":[0.000019407,0.00003833597,0.00010120826,0.000017522083,0.000007078603,0.000013898463,0.000007305587,0.9893406,0.00038013794,0.009809124,0.00026031787,0.0000051508],"about_ca_topic_score_codex":0.0046568294,"about_ca_topic_score_gemma":0.0035494266,"teacher_disagreement_score":0.0070958943,"about_ca_system_score_codex":0.0014262199,"about_ca_system_score_gemma":0.0028207118,"threshold_uncertainty_score":0.037527084},"labels":[],"label_agreement":null},{"id":"W3132089310","doi":"10.1109/iros45743.2020.9341019","title":"Learning Domain Randomization Distributions for Training Robust Locomotion Policies","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Artificial intelligence; Context (archaeology); Machine learning; Task (project management); Generalization; Extrapolation; Domain (mathematical analysis); Interpolation (computer graphics); Range (aeronautics); Mathematics; Statistics; Engineering","score_opus":0.05326927639304079,"score_gpt":0.25799983112925,"score_spread":0.2047305547362092,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3132089310","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.054500632,0.000263107,0.94272465,0.00019858939,0.000026748403,0.00009808207,0.000052907773,0.0009539689,0.0011812916],"genre_scores_gemma":[0.81957215,0.000211771,0.17780313,0.00022342811,0.000036831705,0.0004179436,0.00025677204,0.000183432,0.0012945198],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990006,0.00048802097,0.000052772604,0.000219461,0.00014917023,0.00008998437],"domain_scores_gemma":[0.99322695,0.005200333,0.00053646724,0.0004448537,0.00037369714,0.00021772536],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002775985,0.0010240986,0.0010620655,0.00081576034,0.0003794746,0.00063206645,0.0013302377,0.001320889,0.0016074813],"category_scores_gemma":[0.013637765,0.00078160735,0.0005434365,0.00038523713,0.0014331281,0.0015780793,0.0015735481,0.0021264045,0.00055762305],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010675745,0.00006945739,0.00064844824,0.000035822923,0.000020101183,0.000023896508,0.000036612335,0.97434604,0.0011751705,0.0040039658,0.0002913729,0.019242322],"study_design_scores_gemma":[0.000013073923,0.000040187453,0.00005657905,0.0000064157325,0.0000023781513,0.000005770974,0.000004397776,0.9964142,0.00047867958,0.0028586262,0.0001157429,0.0000039027927],"about_ca_topic_score_codex":0.0016459337,"about_ca_topic_score_gemma":0.001670763,"teacher_disagreement_score":0.002775985,"about_ca_system_score_codex":0.0010183909,"about_ca_system_score_gemma":0.0011101691,"threshold_uncertainty_score":0.014680982},"labels":[],"label_agreement":null},{"id":"W3132255199","doi":"","title":"C-Learning: Horizon-Aware Cumulative Accessibility Estimation","year":2021,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reachability; Reinforcement learning; Computer science; Motion planning; Time horizon; Generalization; Sample (material); Reliability (semiconductor); Set (abstract data type); Path (computing); Code (set theory); Field (mathematics); Machine learning; Artificial intelligence; Mathematical optimization; Robot; Theoretical computer science; Mathematics","score_opus":0.06374053049019669,"score_gpt":0.3772116194799136,"score_spread":0.31347108898971693,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3132255199","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0060521197,0.00015833261,0.9905992,0.00010042794,0.000027063914,0.000056924768,0.000098307566,0.0017258298,0.0011818015],"genre_scores_gemma":[0.5669572,0.00029470216,0.4266716,0.00031363673,0.000095181626,0.000392812,0.0006985349,0.00075465924,0.0038217776],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991186,0.00018572022,0.00004609963,0.00023704306,0.00027202375,0.00014050954],"domain_scores_gemma":[0.9955445,0.0027698022,0.0003366297,0.0005164863,0.00055062847,0.00028200078],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013644099,0.0011856774,0.0015232633,0.0011766148,0.00060328486,0.0011209298,0.0027601963,0.0015709036,0.0059993914],"category_scores_gemma":[0.008942839,0.00059396005,0.0008401045,0.0009341883,0.0011806885,0.0019375188,0.0024481788,0.002578844,0.0009219241],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016530354,0.00011105186,0.0010518226,0.00014034654,0.000043906846,0.000075119664,0.00007619515,0.7847274,0.0016505676,0.012352512,0.0044203303,0.19518544],"study_design_scores_gemma":[0.000012548845,0.000018454299,0.000065872504,0.000008425112,0.000004449879,0.000013211528,0.000004977605,0.9919872,0.0005358824,0.00692613,0.0004175201,0.000005343176],"about_ca_topic_score_codex":0.011520865,"about_ca_topic_score_gemma":0.013147348,"teacher_disagreement_score":0.011520865,"about_ca_system_score_codex":0.0014351611,"about_ca_system_score_gemma":0.0029892002,"threshold_uncertainty_score":0.022907615},"labels":[],"label_agreement":null},{"id":"W3132477017","doi":"","title":"Non-Linear Rewards For Successor Features","year":2021,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Successor cardinal; Computer science; Reinforcement learning; Artificial intelligence; Machine learning; Feature (linguistics); Function (biology); Term (time); Mathematics","score_opus":0.01908450848725854,"score_gpt":0.2887595636198263,"score_spread":0.26967505513256773,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3132477017","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049287166,0.00016828718,0.9477547,0.00026881136,0.00003336848,0.000034624183,0.000060459177,0.00043177104,0.0019608575],"genre_scores_gemma":[0.9380307,0.00009015498,0.058688976,0.000078169505,0.000026612935,0.00008857242,0.00007656113,0.00006151203,0.0028586825],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999283,0.00023617262,0.0000378381,0.00016660917,0.00018145652,0.00009492833],"domain_scores_gemma":[0.99714524,0.0017912547,0.0003363045,0.00032866598,0.0002391559,0.00015943695],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015740613,0.00080448337,0.0009573897,0.00049926754,0.00037406513,0.00082269363,0.001318921,0.00088640273,0.002666224],"category_scores_gemma":[0.0063805133,0.0003924952,0.0006139126,0.00034984027,0.0014814368,0.0022469629,0.0013343666,0.0020132305,0.00036181128],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021835476,0.00014003308,0.0020140347,0.00008039958,0.00003816036,0.0001501834,0.000097692384,0.8469123,0.0024561468,0.07103092,0.0011243328,0.07573744],"study_design_scores_gemma":[0.0000074440486,0.00004169857,0.00017387269,0.000005942037,0.000004283601,0.000021117423,0.0000033931344,0.98537874,0.00037706638,0.013760091,0.00021920889,0.000007116362],"about_ca_topic_score_codex":0.0019251434,"about_ca_topic_score_gemma":0.0024064425,"teacher_disagreement_score":0.002666224,"about_ca_system_score_codex":0.0011183715,"about_ca_system_score_gemma":0.0011255018,"threshold_uncertainty_score":0.008919418},"labels":[],"label_agreement":null},{"id":"W3133261350","doi":"","title":"Watch-And-Help: A Challenge for Social Perception and Human-AI Collaboration","year":2021,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Task (project management); Computer science; Perception; Benchmark (surveying); Social intelligence; Human–computer interaction; Artificial intelligence; Human intelligence; Intelligent agent; Knowledge management; Data science; Psychology; Engineering; Social psychology","score_opus":0.07150864535425586,"score_gpt":0.22592537345845154,"score_spread":0.15441672810419568,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3133261350","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5998609,0.0050425143,0.3343049,0.013299309,0.0013774941,0.001638793,0.0036361217,0.0061275642,0.034712374],"genre_scores_gemma":[0.8681168,0.00060412847,0.121829025,0.0012487146,0.00014544168,0.0007473466,0.003092064,0.0003494842,0.00386703],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9916809,0.0054710167,0.00027995722,0.0011041806,0.0011750938,0.00028883826],"domain_scores_gemma":[0.97936535,0.012620327,0.0010683388,0.0034267986,0.001386469,0.0021325785],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0058276597,0.0012588025,0.00096965727,0.0005045511,0.0012820428,0.0021991911,0.0020758822,0.0029630247,0.003440544],"category_scores_gemma":[0.029192884,0.00040846094,0.00074890564,0.00039213768,0.0027243479,0.0048416904,0.0046915584,0.0027383051,0.0013526859],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031243241,0.005704275,0.031884864,0.0034114658,0.000692414,0.0007740424,0.0061970977,0.26147467,0.025269628,0.05075198,0.09375185,0.5169635],"study_design_scores_gemma":[0.00057695946,0.004151502,0.019252365,0.00035004778,0.00011921741,0.0008361341,0.005247478,0.716012,0.022459619,0.14001152,0.09066201,0.00032120102],"about_ca_topic_score_codex":0.004603507,"about_ca_topic_score_gemma":0.004926256,"teacher_disagreement_score":0.0058276597,"about_ca_system_score_codex":0.001270095,"about_ca_system_score_gemma":0.0014957906,"threshold_uncertainty_score":0.030820012},"labels":[],"label_agreement":null},{"id":"W3134156689","doi":"10.48550/arxiv.2103.00107","title":"Revisiting Peng's Q($\\lambda$) for Modern Reinforcement Learning","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Convergence (economics); Lambda; Algorithm; Function (biology); Computer science; Mathematics; Artificial intelligence; Physics; Economics; Quantum mechanics","score_opus":0.07805414104499778,"score_gpt":0.20661398747231488,"score_spread":0.1285598464273171,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3134156689","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0117915375,0.0003629281,0.98006445,0.0007432278,0.00009480755,0.00006517866,0.0000305946,0.0005625195,0.0062847817],"genre_scores_gemma":[0.5760193,0.00065992604,0.4138805,0.000927549,0.00014806628,0.0002651915,0.00009996811,0.00032939535,0.0076701655],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99889755,0.00042015561,0.00004475046,0.00022664777,0.0003080038,0.000102834565],"domain_scores_gemma":[0.9960699,0.0027685007,0.00016430089,0.00051719864,0.0003391959,0.0001407451],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028669487,0.0008989923,0.00071036635,0.0005990569,0.00075552077,0.0011433689,0.0018315675,0.0012419743,0.003801266],"category_scores_gemma":[0.011600052,0.00037736888,0.0006020964,0.0005286086,0.0026928498,0.0020509767,0.0021181582,0.003610357,0.00083312194],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017419347,0.00014596514,0.0023309975,0.0002292528,0.000056249515,0.000116636205,0.00031925272,0.32295093,0.0028355483,0.50207216,0.0055054026,0.16326341],"study_design_scores_gemma":[0.000031070165,0.00007838148,0.00016133407,0.000034050758,0.000010016883,0.00004155816,0.000017694705,0.8349685,0.0008956626,0.15982811,0.003919704,0.000013856015],"about_ca_topic_score_codex":0.0043620197,"about_ca_topic_score_gemma":0.0036670296,"teacher_disagreement_score":0.0043620197,"about_ca_system_score_codex":0.0016342198,"about_ca_system_score_gemma":0.0025183884,"threshold_uncertainty_score":0.01516211},"labels":[],"label_agreement":null},{"id":"W3134456772","doi":"10.48550/arxiv.2103.03216","title":"Continuous Coordination As a Realistic Scenario for Lifelong Learning","year":2021,"lang":"en","type":"preprint","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Université de Montréal","funders":"","keywords":"Lifelong learning; Computer science; Reinforcement learning; Testbed; Benchmark (surveying); Task (project management); Artificial intelligence; Engineering","score_opus":0.01464353487020289,"score_gpt":0.2543663899315203,"score_spread":0.23972285506131738,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3134456772","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.54582715,0.0009177171,0.42329583,0.0027623475,0.00030392644,0.0006111539,0.0015618318,0.0033781976,0.021341877],"genre_scores_gemma":[0.94977385,0.00011853086,0.046352904,0.00028390397,0.000022434151,0.00032379746,0.0006637217,0.000087590415,0.0023732649],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988738,0.00045052246,0.000047942656,0.000293848,0.00015184737,0.00018198257],"domain_scores_gemma":[0.9977812,0.001104955,0.0001817648,0.0003889011,0.00017075305,0.00037239492],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014015863,0.0006975151,0.0008226583,0.00028787518,0.00085178186,0.0011286105,0.0018951765,0.0015639531,0.003855003],"category_scores_gemma":[0.0053767613,0.00036084917,0.00045963045,0.00024619998,0.0016374484,0.002071017,0.0021427383,0.0020588504,0.0006342684],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051031855,0.00050499773,0.0032711527,0.00020809147,0.00007778458,0.0005739787,0.00032097386,0.9299215,0.004518319,0.035581477,0.006091944,0.018419316],"study_design_scores_gemma":[0.00011688502,0.00019265385,0.0005995283,0.000022314556,0.000009090847,0.00007627579,0.000105623,0.9698635,0.0019089136,0.02450342,0.0025801654,0.000021672751],"about_ca_topic_score_codex":0.004250502,"about_ca_topic_score_gemma":0.004733857,"teacher_disagreement_score":0.004250502,"about_ca_system_score_codex":0.0010543178,"about_ca_system_score_gemma":0.0011531387,"threshold_uncertainty_score":0.01289624},"labels":[],"label_agreement":null},{"id":"W3136163870","doi":"10.48550/arxiv.2011.13897","title":"Latent Skill Planning for Exploration and Transfer","year":2020,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Reinforcement learning; Leverage (statistics); Suite; Task (project management); Adaptation (eye); Amortization; Knowledge transfer; Transfer of learning; Artificial intelligence; Machine learning; Human–computer interaction; Knowledge management","score_opus":0.14792688945634863,"score_gpt":0.19303798547250298,"score_spread":0.045111096016154345,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3136163870","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016030798,0.00031260253,0.9776585,0.00037682653,0.000051190833,0.00013049493,0.00014471951,0.0020120465,0.0032827829],"genre_scores_gemma":[0.7179636,0.00031274994,0.2753499,0.00026031732,0.00007289498,0.0007661228,0.000456865,0.0003736808,0.004443886],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991974,0.0002673613,0.00004560636,0.00022498282,0.00014823758,0.00011627093],"domain_scores_gemma":[0.9978224,0.0011866366,0.00016966827,0.00051525806,0.00014782537,0.0001582318],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001574918,0.0011752454,0.0009842034,0.0005657303,0.00049722224,0.000976889,0.0019614021,0.0010792423,0.008144203],"category_scores_gemma":[0.007074319,0.000610391,0.0007736481,0.0005633842,0.0016178803,0.0020055806,0.0025729416,0.0025475733,0.0013812832],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029414194,0.00028174222,0.0014523123,0.000264951,0.000073748204,0.000102956576,0.00016721513,0.7306595,0.0042217327,0.05733128,0.004112576,0.20103787],"study_design_scores_gemma":[0.000032661053,0.00007233198,0.000136411,0.0000150419255,0.000009884305,0.000015054469,0.000011873049,0.9403905,0.0008420088,0.05751517,0.0009509415,0.000008029907],"about_ca_topic_score_codex":0.0038392195,"about_ca_topic_score_gemma":0.004514314,"teacher_disagreement_score":0.008144203,"about_ca_system_score_codex":0.0015518921,"about_ca_system_score_gemma":0.0024652833,"threshold_uncertainty_score":0.027245045},"labels":[],"label_agreement":null},{"id":"W3136181012","doi":"10.1109/access.2021.3065710","title":"A Hybrid Multi-Task Learning Approach for Optimizing Deep Reinforcement Learning Agents","year":2021,"lang":"en","type":"article","venue":"IEEE Access","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Task (project management); Machine learning; Field (mathematics); Multi-task learning; Deep learning; Engineering","score_opus":0.0610404448592446,"score_gpt":0.314948900322499,"score_spread":0.2539084554632544,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3136181012","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026355129,0.00030219098,0.96955025,0.00024008223,0.000045300403,0.000058583737,0.00002764559,0.00026209652,0.003158801],"genre_scores_gemma":[0.8955366,0.00013101727,0.1006238,0.00019061443,0.00004076266,0.00018886467,0.00005429434,0.00004402546,0.0031899877],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995603,0.0001505938,0.000022453916,0.00008847626,0.000102305225,0.00007583931],"domain_scores_gemma":[0.9992995,0.0003180131,0.00009328041,0.00004148137,0.00018457056,0.00006315799],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012558803,0.0010348922,0.0009894916,0.00040584258,0.000362414,0.0007985412,0.001394824,0.0011819048,0.0017971882],"category_scores_gemma":[0.001859339,0.00042540257,0.0005598499,0.00031979635,0.00071068184,0.0007285876,0.0011552607,0.0010898053,0.0002750637],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000031024225,0.000026447844,0.00021865197,0.000021835345,0.000021564765,0.000034137807,0.000021018821,0.9854939,0.00069765403,0.0021996049,0.00023006341,0.011003964],"study_design_scores_gemma":[0.0000040250466,0.000015172616,0.000017101316,0.0000013398862,0.0000021972298,0.0000029652351,0.0000015381353,0.99929607,0.00008975707,0.00048666066,0.000081667946,0.0000014280586],"about_ca_topic_score_codex":0.004832186,"about_ca_topic_score_gemma":0.0038449778,"teacher_disagreement_score":0.004832186,"about_ca_system_score_codex":0.0007958209,"about_ca_system_score_gemma":0.0011430837,"threshold_uncertainty_score":0.00960809},"labels":[],"label_agreement":null},{"id":"W3136492000","doi":"10.1177/1059712321999421","title":"Affordance as general value function: a computational model","year":2021,"lang":"en","type":"article","venue":"Adaptive Behavior","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Huawei Technologies (Canada)","funders":"","keywords":"Affordance; Computer science; Perception; Reinforcement learning; Artificial intelligence; Action (physics); Function (biology); Cognitive science; Perspective (graphical); Explication; Value (mathematics); Scalability; Human–computer interaction; Machine learning; Psychology; Epistemology","score_opus":0.030329927443571846,"score_gpt":0.2752280964595858,"score_spread":0.24489816901601394,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3136492000","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024186784,0.0013309686,0.9487096,0.002708036,0.000087212116,0.000038048664,0.00037911843,0.00029773498,0.02226256],"genre_scores_gemma":[0.82531744,0.002539676,0.15408933,0.00048026082,0.00017124195,0.00031739473,0.0003320563,0.00018258623,0.016569994],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99973494,0.000086699474,0.000011785771,0.00007773915,0.000054163072,0.00003469661],"domain_scores_gemma":[0.9994647,0.0003133194,0.000060410166,0.00006251107,0.00004667317,0.00005235898],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005690302,0.00062654435,0.0007798225,0.0006319053,0.00043261645,0.0018206346,0.0016825349,0.0017693293,0.0051376903],"category_scores_gemma":[0.0025371476,0.00047639667,0.0012133371,0.00073005067,0.0024389334,0.0042504156,0.0014192044,0.002370953,0.00067035947],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017069244,0.000018649764,0.0003509505,0.000055051027,0.000022615663,0.0000726747,0.00010389293,0.25817654,0.000608452,0.7303839,0.0012055511,0.008984589],"study_design_scores_gemma":[0.0000070430224,0.000016190404,0.00013162686,0.000017376642,0.000007821309,0.000041675532,0.000014788997,0.5596939,0.00009375458,0.4377488,0.0022167552,0.0000103152615],"about_ca_topic_score_codex":0.0043291976,"about_ca_topic_score_gemma":0.002995213,"teacher_disagreement_score":0.0051376903,"about_ca_system_score_codex":0.0010749968,"about_ca_system_score_gemma":0.0008252139,"threshold_uncertainty_score":0.017187238},"labels":[],"label_agreement":null},{"id":"W3136541527","doi":"10.1287/moor.2022.1331","title":"Convergence of Finite Memory Q Learning for POMDPs and Near Optimality of Learned Policies Under Filter Stability","year":2022,"lang":"en","type":"article","venue":"Mathematics of Operations Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Queen's University","funders":"","keywords":"Markov decision process; Partially observable Markov decision process; Convergence (economics); Bellman equation; Mathematical optimization; Mathematics; Limit (mathematics); Stability (learning theory); Filter (signal processing); Q-learning; Reinforcement learning; Optimal control; Ergodicity; Markov chain; Applied mathematics; Computer science; Markov process; Artificial intelligence; Machine learning","score_opus":0.19567375944599658,"score_gpt":0.3990674966197256,"score_spread":0.20339373717372905,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3136541527","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033055406,0.00015793616,0.96442497,0.00028991673,0.000017601476,0.000054227843,0.00003205114,0.00015209922,0.0018157299],"genre_scores_gemma":[0.8718561,0.00034630724,0.12455048,0.00024694492,0.00004490515,0.00039969364,0.00013182206,0.00015088964,0.0022729237],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976433,0.00093060697,0.00014469637,0.00047888194,0.00054735894,0.0002551225],"domain_scores_gemma":[0.9588679,0.035531618,0.0019823005,0.0010794197,0.0019185923,0.00062016054],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007389967,0.0009997521,0.0014132909,0.0010984658,0.0009813897,0.001761536,0.0015003277,0.0019649672,0.0024683503],"category_scores_gemma":[0.0530324,0.00066819554,0.0010827674,0.00056112563,0.0038063363,0.0035742174,0.0027933843,0.0026531543,0.00033468805],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017624303,0.000100093544,0.0018489711,0.00015232866,0.000069113004,0.00013525772,0.00030921792,0.83992827,0.0020572434,0.1370775,0.0004172377,0.01772861],"study_design_scores_gemma":[0.000015076115,0.00004978177,0.00011279247,0.000016471078,0.000004837877,0.000014954669,0.000016212285,0.96443504,0.00065847166,0.034558658,0.00011003252,0.0000075712883],"about_ca_topic_score_codex":0.004182089,"about_ca_topic_score_gemma":0.0019360182,"teacher_disagreement_score":0.007389967,"about_ca_system_score_codex":0.002279695,"about_ca_system_score_gemma":0.0027857055,"threshold_uncertainty_score":0.03908235},"labels":[],"label_agreement":null},{"id":"W3139073295","doi":"10.48550/arxiv.2103.07945","title":"Learning One Representation to Optimize All Rewards","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Representation (politics); Markov decision process; Artificial intelligence; Reinforcement learning; A priori and a posteriori; Range (aeronautics); Machine learning; Markov process; Mathematics","score_opus":0.12253217452495434,"score_gpt":0.22301649940082927,"score_spread":0.10048432487587493,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3139073295","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012637869,0.00015928062,0.983356,0.00042125167,0.000047160604,0.000034631805,0.00015524481,0.000668063,0.002520533],"genre_scores_gemma":[0.68069243,0.0004005587,0.30940142,0.00029784258,0.00010558102,0.00032206922,0.000548801,0.00031751988,0.007913791],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994535,0.00015905067,0.000025040888,0.00017462687,0.00009655444,0.00009132223],"domain_scores_gemma":[0.9991881,0.00035842985,0.00008894473,0.00018276644,0.000107876745,0.00007381566],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008321302,0.0011723958,0.0011304736,0.00050109235,0.00033088817,0.001236038,0.0015103042,0.0015772209,0.004178167],"category_scores_gemma":[0.0037968755,0.0004661395,0.00066226994,0.0005159405,0.0009843919,0.0022803012,0.0015812425,0.0022399994,0.0010473966],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011982171,0.000112310416,0.000606863,0.00009450939,0.000045731664,0.00006606287,0.00006715308,0.85206765,0.0019877646,0.06094719,0.0030313868,0.08085354],"study_design_scores_gemma":[0.000013425292,0.000036318983,0.000049762177,0.0000099106,0.000007680539,0.00001547573,0.0000050377844,0.97040075,0.0006638508,0.02815738,0.0006328491,0.00000758119],"about_ca_topic_score_codex":0.0021028777,"about_ca_topic_score_gemma":0.002484579,"teacher_disagreement_score":0.004178167,"about_ca_system_score_codex":0.0010699742,"about_ca_system_score_gemma":0.0017212017,"threshold_uncertainty_score":0.013977349},"labels":[],"label_agreement":null},{"id":"W3142824457","doi":"10.1142/s2737480721500059","title":"Path Following Control for UAV Using Deep Reinforcement Learning Approach","year":2021,"lang":"en","type":"article","venue":"Guidance Navigation and Control","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Reinforcement learning; Computer science; Path (computing); Convergence (economics); Motion planning; Function (biology); Artificial intelligence; Control (management); Trajectory; Q-learning; Action (physics); Domain (mathematical analysis); Mathematical optimization; Algorithm; Control theory (sociology); Mathematics; Robot","score_opus":0.016363631021404385,"score_gpt":0.25880171570533134,"score_spread":0.24243808468392694,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3142824457","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0404742,0.0004632351,0.9535398,0.00027089464,0.000070215916,0.00003719819,0.00003131434,0.0003823073,0.0047309133],"genre_scores_gemma":[0.9692313,0.00013792983,0.028264333,0.00008178964,0.000014979246,0.00006593213,0.00003564317,0.000019020063,0.0021489873],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998733,0.000022975968,0.0000064259575,0.000031265467,0.00003170212,0.000034279277],"domain_scores_gemma":[0.99973744,0.00010775021,0.000051334813,0.000013144776,0.00006870595,0.000021666781],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003437528,0.0005312825,0.0005065531,0.0002067662,0.00030366716,0.00042583302,0.00059000484,0.00057657075,0.0012705389],"category_scores_gemma":[0.00072937546,0.00022342532,0.0003024046,0.0001673777,0.0004329015,0.00035444717,0.0005894003,0.00074803224,0.00012456567],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000044433546,0.000025398196,0.00044841383,0.00003874336,0.000014022083,0.000057132536,0.00003513287,0.9697798,0.0018518597,0.0032488632,0.0004986616,0.023957523],"study_design_scores_gemma":[0.0000050520584,0.000017493805,0.0000380981,0.0000020111208,0.0000019429147,0.0000047443673,0.0000020148047,0.99919945,0.00015106479,0.00046331744,0.00011356421,0.0000013807213],"about_ca_topic_score_codex":0.011163498,"about_ca_topic_score_gemma":0.007709675,"teacher_disagreement_score":0.011163498,"about_ca_system_score_codex":0.0005928769,"about_ca_system_score_gemma":0.00097263616,"threshold_uncertainty_score":0.022197068},"labels":[],"label_agreement":null},{"id":"W3143003892","doi":"10.3390/app11073068","title":"New Approach in Human-AI Interaction by Reinforcement-Imitation Learning","year":2021,"lang":"en","type":"article","venue":"Applied Sciences","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Reinforcement learning; Leverage (statistics); Computer science; Imitation; Action (physics); Artificial intelligence; Process (computing); Asynchronous communication; Psychology; Social psychology","score_opus":0.032627967572635294,"score_gpt":0.2980415984961594,"score_spread":0.2654136309235241,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3143003892","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0049336865,0.0006502968,0.9879159,0.00039134375,0.00008163377,0.000039084727,0.0000108778295,0.00034777416,0.0056294035],"genre_scores_gemma":[0.54653823,0.001178858,0.4387815,0.00030800686,0.00019506802,0.0002667922,0.000055559343,0.0001313164,0.012544563],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994288,0.00020134983,0.000026716943,0.00013629586,0.00016887947,0.000037952304],"domain_scores_gemma":[0.9995851,0.00019436721,0.00004276564,0.000063957195,0.00007315086,0.000040697778],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008605304,0.0006581546,0.0005535611,0.0004589129,0.00039534352,0.00090190687,0.001500169,0.0010241488,0.0023598482],"category_scores_gemma":[0.0013909533,0.00028089518,0.000617833,0.00028672063,0.0012463738,0.0011355501,0.001265736,0.0011377209,0.0005740386],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013853244,0.00022593685,0.0017089939,0.00042170042,0.0002006016,0.00042813705,0.0007177965,0.456387,0.016471526,0.2663206,0.0041722376,0.25280684],"study_design_scores_gemma":[0.000024312763,0.000094606585,0.00017867071,0.000020279947,0.000020332584,0.000110177396,0.00003141775,0.9542393,0.0018908472,0.034059793,0.009312705,0.00001755373],"about_ca_topic_score_codex":0.0015182139,"about_ca_topic_score_gemma":0.0011184195,"teacher_disagreement_score":0.0023598482,"about_ca_system_score_codex":0.0005219832,"about_ca_system_score_gemma":0.00071263756,"threshold_uncertainty_score":0.007894456},"labels":[],"label_agreement":null},{"id":"W3144252049","doi":"","title":"RL Unplugged: A Collection of Benchmarks for Offline Reinforcement Learning.","year":2020,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Machine learning","score_opus":0.025197425025171277,"score_gpt":0.25185865706341515,"score_spread":0.22666123203824387,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3144252049","genre_codex":"methods","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25822547,0.010340473,0.57605386,0.0025785086,0.0020674479,0.0015256127,0.027346939,0.036556523,0.08530521],"genre_scores_gemma":[0.7069229,0.0010972391,0.252745,0.00053404324,0.00017330206,0.0013420206,0.024854166,0.002749137,0.0095822355],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9967495,0.0013005846,0.00027536828,0.00052708277,0.00080430065,0.0003431504],"domain_scores_gemma":[0.98730487,0.0078604035,0.0005483033,0.0021046365,0.0015595439,0.0006221647],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004315979,0.002597443,0.0012483287,0.0016273971,0.0008238927,0.0015683688,0.0036128773,0.0020245153,0.0066732448],"category_scores_gemma":[0.026208658,0.0005629679,0.0009539207,0.0012546058,0.0010975945,0.0016971058,0.0019830812,0.003357156,0.0022817662],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017206131,0.001788027,0.0035894734,0.0014438935,0.0003040172,0.00028806107,0.00016592482,0.70386577,0.002863432,0.014413037,0.078878716,0.19067909],"study_design_scores_gemma":[0.0002369867,0.0004212072,0.0012140871,0.000089197594,0.000034857465,0.0000906433,0.00005234592,0.9704902,0.0037772348,0.016953329,0.006608001,0.00003190621],"about_ca_topic_score_codex":0.010854483,"about_ca_topic_score_gemma":0.01664212,"teacher_disagreement_score":0.010854483,"about_ca_system_score_codex":0.0018338846,"about_ca_system_score_gemma":0.0024028919,"threshold_uncertainty_score":0.0228253},"labels":[],"label_agreement":null},{"id":"W3152801112","doi":"10.48550/arxiv.2104.08543","title":"Planning with Expectation Models for Control","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Reinforcement learning; Variance (accounting); Computer science; Bellman equation; Function (biology); Feature (linguistics); Artificial intelligence; Sample (material); Action (physics); Stochastic modelling; Control (management); Mathematical optimization; Mathematics; Statistics","score_opus":0.0911888882141892,"score_gpt":0.19578933152223352,"score_spread":0.10460044330804431,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3152801112","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0029458876,0.0007101811,0.99121255,0.0005923224,0.00004844725,0.000035256988,0.00006013693,0.00024558045,0.004149618],"genre_scores_gemma":[0.59187984,0.0019938059,0.39641163,0.00069817406,0.0002338502,0.00059325324,0.0003515573,0.00024762476,0.0075903176],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99843305,0.000712399,0.000067783156,0.00032778998,0.0003350082,0.00012392159],"domain_scores_gemma":[0.99351454,0.0053194542,0.00030734623,0.0003747282,0.0003483998,0.00013546825],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025459381,0.0017967091,0.0013705173,0.00064843195,0.0005846499,0.0019539448,0.0015379245,0.0017385216,0.006516162],"category_scores_gemma":[0.011479231,0.00071042724,0.0012116712,0.00096334127,0.002099628,0.003432924,0.002126971,0.0044067837,0.00095337315],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000113737355,0.00008351412,0.00052629196,0.00016833756,0.000052827745,0.000073405645,0.0001372929,0.6711638,0.00055221084,0.27974632,0.0025729511,0.044809338],"study_design_scores_gemma":[0.000021190117,0.000038929753,0.000054663502,0.000025043011,0.000008874553,0.000017661143,0.0000123594145,0.82216483,0.00027187186,0.17576218,0.0016092485,0.000013110562],"about_ca_topic_score_codex":0.0061477628,"about_ca_topic_score_gemma":0.0044552926,"teacher_disagreement_score":0.006516162,"about_ca_system_score_codex":0.0025042994,"about_ca_system_score_gemma":0.0019565662,"threshold_uncertainty_score":0.02179873},"labels":[],"label_agreement":null},{"id":"W3153326893","doi":"","title":"Approximate Multi-Agent Fitted Q Iteration.","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computation; Reinforcement learning; Mathematical optimization; Computer science; Function (biology); Property (philosophy); Mathematics; Applied mathematics; Algorithm; Artificial intelligence","score_opus":0.09705536036995745,"score_gpt":0.20003257660552953,"score_spread":0.10297721623557209,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3153326893","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0037505305,0.00008651509,0.9946485,0.0000915059,0.000029397093,0.000051968116,0.000018311233,0.00019868478,0.0011246989],"genre_scores_gemma":[0.5351829,0.00015281777,0.45985022,0.00027846472,0.00005766383,0.00041278315,0.0001412673,0.00012840156,0.0037954631],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988674,0.00047270453,0.000050745653,0.00017439119,0.0002975913,0.00013722079],"domain_scores_gemma":[0.9970222,0.0016367759,0.00031544734,0.0003321173,0.0004987411,0.00019479013],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002392458,0.0008471334,0.0011671114,0.00045082756,0.00045071216,0.000924882,0.0020879344,0.0016331271,0.0032522022],"category_scores_gemma":[0.008767345,0.00051955215,0.00067248166,0.00043928315,0.0013331604,0.0012891982,0.0013648071,0.0016799929,0.0009312921],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001103531,0.00005638925,0.00067328283,0.0000663864,0.00003129663,0.00005524745,0.000056263714,0.94406146,0.00092246477,0.023900883,0.0011170025,0.028949074],"study_design_scores_gemma":[0.000007251181,0.000016912838,0.000023554043,0.0000032822195,0.000001925053,0.000007720263,0.0000028447807,0.995952,0.00017181756,0.0035404242,0.00027030002,0.0000020858918],"about_ca_topic_score_codex":0.005103405,"about_ca_topic_score_gemma":0.004273933,"teacher_disagreement_score":0.005103405,"about_ca_system_score_codex":0.0013083963,"about_ca_system_score_gemma":0.002683048,"threshold_uncertainty_score":0.012652695},"labels":[],"label_agreement":null},{"id":"W3156545714","doi":"10.1177/10597123221095880","title":"What’s a good prediction? Challenges in evaluating an agent’s knowledge","year":2022,"lang":"en","type":"article","venue":"Adaptive Behavior","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates; University of Alberta; Canada Research Chairs; DeepMind; Alberta Machine Intelligence Institute; Canadian Institute for Advanced Research","keywords":"Computer science; Artificial intelligence; Knowledge management; Cognitive science; Machine learning; Psychology","score_opus":0.19113708756580008,"score_gpt":0.36126874226963157,"score_spread":0.1701316547038315,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3156545714","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4040055,0.0063686315,0.4906971,0.046940234,0.00057681877,0.00043746628,0.00077552814,0.001738384,0.04846025],"genre_scores_gemma":[0.9137957,0.00064943347,0.0837396,0.00072696275,0.00008723074,0.000085146356,0.0001667441,0.000121158264,0.00062795234],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.981385,0.01093605,0.0009540912,0.0018507327,0.0044066356,0.00046750228],"domain_scores_gemma":[0.906543,0.07098107,0.0048565804,0.0072282315,0.008035792,0.0023553397],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.025073253,0.0008423746,0.0016551708,0.0011992109,0.0011123132,0.006689593,0.0019064323,0.0031647154,0.0019222368],"category_scores_gemma":[0.1164608,0.00033659756,0.0005097783,0.001080799,0.0035899794,0.0105794035,0.0019700187,0.0034601188,0.0006112986],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012764251,0.0010275915,0.06656356,0.0012869908,0.0006586082,0.00057708775,0.0031475744,0.1714058,0.0064495616,0.13711622,0.015328317,0.5951622],"study_design_scores_gemma":[0.00016169771,0.0009414081,0.017884057,0.00088847184,0.00017504343,0.00034007212,0.0028443546,0.5539283,0.008893692,0.3987759,0.014945226,0.00022174293],"about_ca_topic_score_codex":0.0034207236,"about_ca_topic_score_gemma":0.003577757,"teacher_disagreement_score":0.025073253,"about_ca_system_score_codex":0.0016247635,"about_ca_system_score_gemma":0.001929067,"threshold_uncertainty_score":0.13260156},"labels":[],"label_agreement":null},{"id":"W3157293568","doi":"10.48550/arxiv.2104.13877","title":"Autoregressive Dynamics Models for Offline Policy Evaluation and\\n Optimization","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Autoregressive model; Computer science; Conditional independence; Covariance; Feed forward; Artificial intelligence; Econometrics; Mathematics; Statistics","score_opus":0.08714489944938392,"score_gpt":0.2331941443887187,"score_spread":0.14604924493933477,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3157293568","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006408984,0.00046542136,0.9897523,0.00029368178,0.00003919018,0.00003308079,0.00011582348,0.0010646399,0.0018269329],"genre_scores_gemma":[0.7065876,0.0009643268,0.28185508,0.00039926823,0.00014125922,0.00043983004,0.0007999225,0.00056082045,0.008251959],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991768,0.0003376791,0.000046203197,0.00017612253,0.00018034661,0.000082858984],"domain_scores_gemma":[0.9970208,0.0023347244,0.00016790109,0.0001805674,0.00022575576,0.000070306276],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016216934,0.0013378953,0.0015075218,0.0006257014,0.00037871947,0.0014067844,0.0014749763,0.0012393355,0.004173945],"category_scores_gemma":[0.007067887,0.00084495515,0.0007010391,0.0006641016,0.0009200646,0.0017362465,0.0011346196,0.0032229468,0.0010431708],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000046850175,0.000038830123,0.0002488143,0.000052363284,0.00002519854,0.00002719581,0.00002514477,0.95718133,0.00036443834,0.011855012,0.0009505927,0.029184299],"study_design_scores_gemma":[0.0000029277485,0.0000060397088,0.000021448332,0.0000040339823,0.000002149997,0.0000020178484,0.0000017033052,0.99592376,0.000118091564,0.003718619,0.00019702839,0.0000021690678],"about_ca_topic_score_codex":0.014876941,"about_ca_topic_score_gemma":0.014832618,"teacher_disagreement_score":0.014876941,"about_ca_system_score_codex":0.0016525807,"about_ca_system_score_gemma":0.0024056186,"threshold_uncertainty_score":0.029580653},"labels":[],"label_agreement":null},{"id":"W3157951743","doi":"10.1109/iros51168.2021.9636440","title":"Seeing All the Angles: Learning Multiview Manipulation Policies for Contact-Rich Tasks from Demonstrations","year":2021,"lang":"en","type":"article","venue":"2021 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Viewpoints; Computer science; Task (project management); Artificial intelligence; Perspective (graphical); Human–computer interaction; Robot; Variety (cybernetics); Computer vision","score_opus":0.11316112666620681,"score_gpt":0.3296329576508928,"score_spread":0.21647183098468598,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3157951743","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16437043,0.00034551302,0.83308685,0.00023004433,0.000026425378,0.00007563468,0.00007912323,0.0006906469,0.001095348],"genre_scores_gemma":[0.942302,0.00010588301,0.05644787,0.00009095153,0.0000157559,0.00008766147,0.0001114861,0.000053793447,0.00078455237],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99957794,0.00012759423,0.000024437722,0.0001239777,0.00008997971,0.000056076362],"domain_scores_gemma":[0.9975923,0.0015527578,0.0003304956,0.00021575604,0.00014110668,0.00016752616],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014773309,0.0006998122,0.0008556784,0.00035514377,0.0002429918,0.00056252757,0.0010862035,0.00077930297,0.0011183268],"category_scores_gemma":[0.005604007,0.00047068507,0.00035878315,0.00026276565,0.00079065596,0.0010843597,0.0011666407,0.0012044378,0.00025023113],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041083852,0.000194019,0.0033018477,0.000091753376,0.00006139725,0.00009266997,0.0001493708,0.88844484,0.008166262,0.003328541,0.00063813856,0.09512033],"study_design_scores_gemma":[0.000020602916,0.000083017956,0.0003635237,0.000007792507,0.0000048926645,0.000014129932,0.0000117552645,0.9954426,0.0014999923,0.0023946196,0.00015038445,0.000006664],"about_ca_topic_score_codex":0.0033704978,"about_ca_topic_score_gemma":0.0032176434,"teacher_disagreement_score":0.0033704978,"about_ca_system_score_codex":0.000698257,"about_ca_system_score_gemma":0.0009297078,"threshold_uncertainty_score":0.007812977},"labels":[],"label_agreement":null},{"id":"W3158681970","doi":"10.48550/arxiv.2104.13844","title":"A Generalized Projected Bellman Error for Off-policy Value Estimation in Reinforcement Learning","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Bellman equation; Temporal difference learning; Function approximation; Hyperparameter; Nonlinear system; Mean squared error; Artificial neural network; Mathematical optimization; Function (biology); Computer science; Approximation error; Linear approximation; Value (mathematics); Mathematics; Applied mathematics; Algorithm; Artificial intelligence; Machine learning; Statistics","score_opus":0.07410652224032353,"score_gpt":0.2396721545053905,"score_spread":0.165565632265067,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3158681970","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002513115,0.00015662985,0.9962668,0.00014270071,0.000041710136,0.000026972933,0.000020127316,0.000112870795,0.00071918877],"genre_scores_gemma":[0.34323794,0.00055513234,0.65062714,0.00043340275,0.00018109704,0.00048039007,0.00023924939,0.00036698603,0.0038785671],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9963097,0.0016405256,0.00021661351,0.00068455073,0.0009374554,0.0002111131],"domain_scores_gemma":[0.99014825,0.0068025147,0.0006648773,0.0008706296,0.0012140634,0.00029968223],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068332395,0.0017174345,0.0016169595,0.0008523189,0.0005568887,0.0018114167,0.0021489963,0.0023333924,0.0031832552],"category_scores_gemma":[0.02812581,0.00069571065,0.00091636996,0.0008781934,0.002995707,0.003607173,0.0032465237,0.0041423156,0.0005816563],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011204152,0.000056263005,0.0007847711,0.00017325729,0.00007184216,0.00005530126,0.00010058509,0.8121724,0.001838457,0.12547731,0.0015867476,0.05757104],"study_design_scores_gemma":[0.0000092284745,0.000036646128,0.000089077315,0.000023915894,0.0000064288265,0.0000152015755,0.0000047540734,0.97188705,0.0006453589,0.026757566,0.00051381416,0.000010929265],"about_ca_topic_score_codex":0.003002718,"about_ca_topic_score_gemma":0.002302143,"teacher_disagreement_score":0.0068332395,"about_ca_system_score_codex":0.0021241524,"about_ca_system_score_gemma":0.0032566492,"threshold_uncertainty_score":0.036138058},"labels":[],"label_agreement":null},{"id":"W3158744994","doi":"","title":"Adaptive Approximate Policy Iteration","year":2021,"lang":"en","type":"article","venue":"International Conference on Artificial Intelligence and Statistics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Regret; Reinforcement learning; Markov decision process; Computer science; Ergodic theory; Tilde; Q-learning; Bellman equation; Mathematical optimization; Novelty; Function (biology); Upper and lower bounds; Function approximation; Scheme (mathematics); Adaptive learning; Online learning; Markov process; Artificial intelligence; Mathematics; Machine learning; Discrete mathematics; Statistics","score_opus":0.12021544814826839,"score_gpt":0.3511960815700123,"score_spread":0.2309806334217439,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3158744994","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010197744,0.00021610045,0.9856376,0.0001479768,0.00005007291,0.000049710212,0.000029186054,0.00043652448,0.003235134],"genre_scores_gemma":[0.7129414,0.00024649155,0.28133783,0.00023167556,0.00007539516,0.00031802332,0.0001497414,0.00014169878,0.004557762],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986993,0.00046382938,0.00006313414,0.00023169396,0.00039051127,0.00015155005],"domain_scores_gemma":[0.9966659,0.0021158766,0.0002580986,0.00038585524,0.0004205177,0.00015383605],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001922425,0.000904548,0.0015063938,0.0005628992,0.00045492177,0.0011484108,0.0017532787,0.0012838452,0.0028093155],"category_scores_gemma":[0.00905413,0.0005088448,0.0005976161,0.0006327258,0.0012886525,0.0013122658,0.0017756772,0.0017711552,0.0005629229],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013253544,0.000068234585,0.0005937805,0.000067042056,0.000038389793,0.000051879553,0.0000657777,0.90023994,0.00079923763,0.0366219,0.0012757787,0.060045525],"study_design_scores_gemma":[0.0000071352874,0.0000158301,0.000021521157,0.0000038727126,0.0000023258167,0.000009556359,0.0000023281293,0.9938167,0.00021595242,0.005627007,0.00027531266,0.0000024223725],"about_ca_topic_score_codex":0.0031686618,"about_ca_topic_score_gemma":0.0024141395,"teacher_disagreement_score":0.0031686618,"about_ca_system_score_codex":0.001199381,"about_ca_system_score_gemma":0.0021018733,"threshold_uncertainty_score":0.0101668835},"labels":[],"label_agreement":null},{"id":"W3159115098","doi":"10.3758/s13415-021-00902-z","title":"Correction to: A win-win situation: Does familiarity with a social robot modulate feedback monitoring and learning?","year":2021,"lang":"en","type":"article","venue":"Cognitive Affective & Behavioral Neuroscience","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Robot; Psychology; Computer science; Artificial intelligence","score_opus":0.02750281265872729,"score_gpt":0.29861624191581393,"score_spread":0.27111342925708665,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3159115098","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00051288697,0.0005716227,0.00074292894,0.10105,0.8943941,0.000047991438,0.0008313549,0.00036109652,0.0014880381],"genre_scores_gemma":[0.06607289,0.004562626,0.006694989,0.16064739,0.59511447,0.0007860062,0.0017705763,0.0011007071,0.1632504],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9974268,0.0003460124,0.0005239161,0.0005704789,0.00080448,0.00032833926],"domain_scores_gemma":[0.97673506,0.007578189,0.0013503998,0.0015076527,0.011236565,0.0015921525],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0024497432,0.002364036,0.0032524574,0.0019645847,0.003763704,0.0030862382,0.004187321,0.016822778,0.07119164],"category_scores_gemma":[0.060664926,0.0011913283,0.0014981205,0.001483599,0.003991241,0.002526226,0.0023480903,0.015814273,0.031066753],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019227709,0.000015200124,0.00017759962,0.00020739416,0.00003386511,0.0013556739,0.000068612666,0.000098035875,0.00016043779,0.000935891,0.9906455,0.00610949],"study_design_scores_gemma":[0.0002477815,0.000072735515,0.0034322212,0.00042132335,0.00008060799,0.0034750064,0.00051875826,0.0016871371,0.0009887621,0.0052331504,0.9836838,0.00015875704],"about_ca_topic_score_codex":0.010414151,"about_ca_topic_score_gemma":0.012921276,"teacher_disagreement_score":0.99755025,"about_ca_system_score_codex":0.005286196,"about_ca_system_score_gemma":0.0032886253,"threshold_uncertainty_score":0.23815978},"labels":[],"label_agreement":null},{"id":"W3160261777","doi":"10.1016/j.automatica.2021.109693","title":"On the convergence of reinforcement learning with Monte Carlo Exploring Starts","year":2021,"lang":"en","type":"article","venue":"Automatica","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; European Commission; Ontario Ministry of Research, Innovation and Science; Canada Research Chairs; Government of Canada; Pacific Institute for the Mathematical Sciences","keywords":"Reinforcement learning; Mathematical optimization; Convergence (economics); Bellman equation; Monte Carlo method; Computer science; Markov decision process; Complement (music); Function (biology); Mathematics; Markov process; Artificial intelligence; Economics","score_opus":0.035769155301195114,"score_gpt":0.22203931889189224,"score_spread":0.18627016359069712,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3160261777","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026467154,0.0010868764,0.9608256,0.0007551044,0.000097546865,0.0000826936,0.00004849933,0.00025232983,0.010384256],"genre_scores_gemma":[0.796362,0.0015100077,0.18978377,0.00050954544,0.0002567123,0.00054056733,0.00022968413,0.00069516274,0.010112668],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9959268,0.0026703684,0.00013325039,0.00035226176,0.0005759687,0.00034146078],"domain_scores_gemma":[0.9170799,0.07612112,0.0015813321,0.0014380811,0.0024996696,0.001279769],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010816423,0.0018558865,0.0031899712,0.0020768938,0.0012229361,0.0020500624,0.003037529,0.0028018993,0.006262085],"category_scores_gemma":[0.064814,0.0016450298,0.0017673722,0.0013205134,0.005905518,0.004203607,0.005131861,0.0051557035,0.00067827967],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023219937,0.00005901518,0.0006578363,0.0001428269,0.00007764349,0.000049101527,0.0001668851,0.8439585,0.0003456118,0.14032257,0.0010261709,0.012961648],"study_design_scores_gemma":[0.000024047868,0.00003314058,0.000064769134,0.000030795705,0.00001007205,0.00000887341,0.000010086554,0.9442423,0.000098066324,0.055280596,0.0001884953,0.00000872812],"about_ca_topic_score_codex":0.008811167,"about_ca_topic_score_gemma":0.004314342,"teacher_disagreement_score":0.010816423,"about_ca_system_score_codex":0.0025204998,"about_ca_system_score_gemma":0.0027392989,"threshold_uncertainty_score":0.057203412},"labels":[],"label_agreement":null},{"id":"W3160504906","doi":"10.35840/2631-5106/4131","title":"Curriculum-Based Deep Reinforcement Learning for Adaptive Robotics: A Mini-Review","year":2021,"lang":"en","type":"article","venue":"International Journal of Robotic Engineering","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia, Okanagan Campus; University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Artificial intelligence; Robotics; Computer science; Automation; Autonomous learning; Autonomous agent; Deep learning; Machine learning; Human–computer interaction; Robot; Engineering; Psychology","score_opus":0.014732341695832863,"score_gpt":0.26142427901113857,"score_spread":0.2466919373153057,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3160504906","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0016592933,0.9473093,0.044443358,0.0014286743,0.0007133524,0.00003757499,0.00006588845,0.00016521664,0.0041772835],"genre_scores_gemma":[0.038627177,0.93630934,0.018904475,0.0007755538,0.0011025828,0.000094307514,0.00022438922,0.00009221675,0.0038698765],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99976057,0.000050468392,0.000027850328,0.00006495435,0.00007339824,0.000022746563],"domain_scores_gemma":[0.998841,0.0008168519,0.000068106965,0.000035241424,0.00019261497,0.00004619463],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011478958,0.0009178511,0.0011085907,0.0006249907,0.00014435494,0.0009837403,0.0017554348,0.0012780598,0.004437408],"category_scores_gemma":[0.0028849028,0.00043738988,0.0006647257,0.0009301357,0.00058910664,0.001775014,0.0009178887,0.0015076661,0.0015622237],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000055095847,0.00008870518,0.00039976859,0.005107078,0.0001417559,0.000045226658,0.000042472293,0.017332787,0.0006469755,0.01082296,0.01118646,0.95413065],"study_design_scores_gemma":[0.000105557825,0.00078076834,0.0032654372,0.010686335,0.0007285853,0.0010120977,0.00018402557,0.1396583,0.006001612,0.051107235,0.7862631,0.00020701603],"about_ca_topic_score_codex":0.0021302234,"about_ca_topic_score_gemma":0.0021745397,"teacher_disagreement_score":0.004437408,"about_ca_system_score_codex":0.00081461266,"about_ca_system_score_gemma":0.0012377116,"threshold_uncertainty_score":0.014844656},"labels":[],"label_agreement":null},{"id":"W3162022061","doi":"10.48550/arxiv.2105.08692","title":"Coach-Player Multi-Agent Reinforcement Learning for Dynamic Team Composition","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Toronto","funders":"","keywords":"Computer science; Reinforcement learning; Task (project management); Composition (language); Generalization; Resource (disambiguation); Team composition; Artificial intelligence; Human–computer interaction; Distributed computing; Knowledge management; Mathematics; Engineering","score_opus":0.06769760757836442,"score_gpt":0.21858040990177732,"score_spread":0.1508828023234129,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3162022061","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023443073,0.00017381385,0.97433734,0.0001846736,0.00003976296,0.000043233886,0.000016615997,0.0003501843,0.0014112064],"genre_scores_gemma":[0.90401,0.00008840458,0.09295513,0.00014109282,0.00003503016,0.00014926051,0.00006370119,0.000078449804,0.002478908],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993831,0.00025135148,0.000024465253,0.0001448593,0.00010919799,0.00008702029],"domain_scores_gemma":[0.99788034,0.0012762765,0.00019308821,0.00017602187,0.0002667096,0.00020764075],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019258673,0.000989585,0.0012572908,0.000398651,0.00058166403,0.00063290674,0.0019384956,0.0012721441,0.0018868983],"category_scores_gemma":[0.0049457042,0.00050309696,0.00046075345,0.00033008948,0.0011886338,0.0009710815,0.0015875552,0.0018468634,0.00036083662],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000070099726,0.000068752655,0.0005967183,0.000032080894,0.0000379284,0.000040819534,0.00005198198,0.9723136,0.000710169,0.005318451,0.00056624314,0.020193184],"study_design_scores_gemma":[0.0000046266923,0.0000089770965,0.00001861013,0.0000011098334,0.0000013462773,0.000002649798,0.0000022028519,0.99856985,0.00008689378,0.0012316826,0.00007082208,0.0000011064462],"about_ca_topic_score_codex":0.0059749414,"about_ca_topic_score_gemma":0.004377205,"teacher_disagreement_score":0.0059749414,"about_ca_system_score_codex":0.0010442093,"about_ca_system_score_gemma":0.0012535219,"threshold_uncertainty_score":0.011880338},"labels":[],"label_agreement":null},{"id":"W3164770271","doi":"10.1007/s00521-021-06104-5","title":"Lucid dreaming for experience replay: refreshing past states with the current policy","year":2021,"lang":"en","type":"article","venue":"Neural Computing and Applications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Army Research Office; Defense Advanced Research Projects Agency; Office of Naval Research; Future of Life Institute; Alberta Machine Intelligence Institute; Natural Sciences and Engineering Research Council of Canada; Lockheed Martin; Robert Bosch (Australia) Pty; Canadian Institute for Advanced Research; National Science Foundation","keywords":"Computer science; State (computer science); Reinforcement learning; Work (physics); Dream; Artificial intelligence; Programming language; Psychology","score_opus":0.018890258616053722,"score_gpt":0.30805375499472293,"score_spread":0.2891634963786692,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3164770271","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17798957,0.0018438644,0.7966172,0.0040978733,0.0008164892,0.00019735593,0.00036991466,0.0047929925,0.013274705],"genre_scores_gemma":[0.93609256,0.00022454618,0.060491163,0.00034599897,0.00004813153,0.00008203987,0.00014829598,0.00015506538,0.0024121497],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995758,0.0001624072,0.000026881924,0.00010745768,0.00006903385,0.00005842451],"domain_scores_gemma":[0.9980896,0.0010150893,0.00012474321,0.0003855822,0.00018778807,0.00019727844],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013716109,0.0006774157,0.0007122882,0.00027245577,0.00045296372,0.0011267205,0.0013856554,0.0009387459,0.0052731563],"category_scores_gemma":[0.011997783,0.00044169778,0.00039326673,0.00021211721,0.0009651493,0.002897227,0.0020644767,0.002254892,0.0007225362],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004215343,0.0009219866,0.008112253,0.00064686965,0.00038827118,0.0008655676,0.0029534944,0.3791716,0.02036084,0.05710418,0.026550313,0.4987092],"study_design_scores_gemma":[0.00011937998,0.0002706665,0.000726485,0.00006115089,0.00006458907,0.00010574927,0.0002636356,0.93991077,0.004130339,0.051006325,0.0032941226,0.00004673298],"about_ca_topic_score_codex":0.0028206836,"about_ca_topic_score_gemma":0.0036096638,"teacher_disagreement_score":0.0052731563,"about_ca_system_score_codex":0.0003895025,"about_ca_system_score_gemma":0.0008194259,"threshold_uncertainty_score":0.017640471},"labels":[],"label_agreement":null},{"id":"W3165914412","doi":"","title":"Pretraining Reward-Free Representations for Data-Efficient Reinforcement Learning","year":2021,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"HEC Montréal; Université de Montréal","funders":"","keywords":"Reinforcement learning; Computer science; Reinforcement; Artificial intelligence; Cognitive psychology; Machine learning; Human–computer interaction; Psychology; Social psychology","score_opus":0.12765188971660188,"score_gpt":0.378776401532506,"score_spread":0.2511245118159041,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3165914412","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02430886,0.00018033192,0.9714717,0.00028466294,0.00007441799,0.00007238128,0.00014933453,0.002066538,0.0013917047],"genre_scores_gemma":[0.7672386,0.00013320919,0.22836173,0.0002683761,0.000048523365,0.0003248257,0.00056215236,0.00024915108,0.0028133984],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99940085,0.00015340293,0.000039041657,0.0001412357,0.00015590608,0.00010953034],"domain_scores_gemma":[0.99707294,0.0017261792,0.00016997925,0.00047513456,0.00043992177,0.00011586581],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011779441,0.0010139418,0.0011583484,0.00051342696,0.00038011384,0.0009448378,0.0019280445,0.0016558936,0.0052073495],"category_scores_gemma":[0.008243588,0.00076832034,0.0005442366,0.0005792119,0.00083278626,0.0021350028,0.0018754463,0.00404458,0.0011637527],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037618767,0.00035368287,0.001110588,0.00015262091,0.00006206656,0.00009968751,0.00008512425,0.6467198,0.007822943,0.01610433,0.0056011807,0.32151172],"study_design_scores_gemma":[0.000018022505,0.00003168353,0.000053994478,0.0000067837136,0.000004329706,0.000011417715,0.000004878778,0.9916238,0.0016284819,0.0063295937,0.00028269828,0.0000041981953],"about_ca_topic_score_codex":0.0038885986,"about_ca_topic_score_gemma":0.0058513,"teacher_disagreement_score":0.0052073495,"about_ca_system_score_codex":0.0009174573,"about_ca_system_score_gemma":0.0018295534,"threshold_uncertainty_score":0.017420352},"labels":[],"label_agreement":null},{"id":"W3165994454","doi":"","title":"Reinforcement Learning as One Big Sequence Modeling Problem","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Sequence (biology); Transformer; Artificial intelligence; Markov chain; Sequence learning; Markov decision process; Machine learning; Markov process; Mathematics; Engineering","score_opus":0.1422045581090734,"score_gpt":0.21029973663436013,"score_spread":0.06809517852528674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3165994454","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016312437,0.00015436573,0.97957695,0.00081294694,0.000037657075,0.000044986613,0.000088257846,0.00051142194,0.002460997],"genre_scores_gemma":[0.8085867,0.00029981128,0.18409087,0.0004755641,0.00007291329,0.00022662971,0.00027430488,0.00018358811,0.0057896213],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99872893,0.00054251414,0.000053196134,0.00036336182,0.00022344525,0.00008851478],"domain_scores_gemma":[0.9968112,0.002314609,0.00018948902,0.00034928764,0.00018749692,0.0001478063],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022999276,0.0008299897,0.0012416283,0.00027969398,0.000362857,0.001139674,0.0015503605,0.0013885306,0.0041029444],"category_scores_gemma":[0.006841409,0.0005032245,0.00076374115,0.00037792875,0.0016739117,0.0029650186,0.0015597935,0.0029698506,0.00049979053],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017290145,0.00007609125,0.0008311134,0.0001219358,0.000050798477,0.00011375687,0.0001745689,0.81033814,0.0017346499,0.13324529,0.001891736,0.051249],"study_design_scores_gemma":[0.000014840798,0.000027668282,0.0000713896,0.000005644073,0.000006675048,0.000016080401,0.00001039144,0.9383397,0.00040522797,0.060495418,0.0006004963,0.0000064716037],"about_ca_topic_score_codex":0.0028833598,"about_ca_topic_score_gemma":0.0030279602,"teacher_disagreement_score":0.0041029444,"about_ca_system_score_codex":0.0012636887,"about_ca_system_score_gemma":0.0013638488,"threshold_uncertainty_score":0.013725698},"labels":[],"label_agreement":null},{"id":"W3167624337","doi":"","title":"Sparse Feature Selection Makes Batch Reinforcement Learning More Sample Efficient","year":2021,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Minimax; Estimator; Reinforcement learning; Lasso (programming language); Upper and lower bounds; Covariance; Divergence (linguistics); Computer science; Sample size determination; Feature selection; Mathematical optimization; Dimension (graph theory); Mathematics; Function (biology); Model selection; Algorithm; Artificial intelligence; Statistics","score_opus":0.030037715410598808,"score_gpt":0.29334601178244984,"score_spread":0.26330829637185105,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3167624337","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007920052,0.00007753014,0.99064845,0.00018447233,0.000022009881,0.00003686429,0.00001949047,0.00031080423,0.0007802336],"genre_scores_gemma":[0.66103405,0.0001567147,0.33478862,0.00040053588,0.00010437318,0.00032669125,0.00017042624,0.00031495286,0.0027036432],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981766,0.000674391,0.000086366344,0.00038952447,0.000513034,0.000160088],"domain_scores_gemma":[0.9912424,0.0061943075,0.0006494965,0.0009898276,0.00068596186,0.00023798084],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004051093,0.0009791417,0.0016404962,0.00037908973,0.00048391524,0.001129911,0.0015123218,0.0012925011,0.0023949633],"category_scores_gemma":[0.018810036,0.00067611085,0.00060487824,0.00039774072,0.001954983,0.0021390193,0.0018588073,0.0027767362,0.0005918536],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021742308,0.00015241123,0.0009786143,0.00008564788,0.000049367227,0.00006185281,0.0000725378,0.904862,0.0044059865,0.033182137,0.0016561637,0.054275885],"study_design_scores_gemma":[0.000011450589,0.000030165607,0.00008144597,0.0000046523214,0.0000039337797,0.000008192198,0.000003467366,0.9912151,0.0006456783,0.007745363,0.00024636058,0.0000041800336],"about_ca_topic_score_codex":0.0026825594,"about_ca_topic_score_gemma":0.0023158102,"teacher_disagreement_score":0.004051093,"about_ca_system_score_codex":0.0011958259,"about_ca_system_score_gemma":0.001957601,"threshold_uncertainty_score":0.021424472},"labels":[],"label_agreement":null},{"id":"W3167763832","doi":"10.21428/594757db.8472938b","title":"General Deep Reinforcement Learning in NES Games","year":2021,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Acadia University","funders":"","keywords":"Reinforcement learning; Domain (mathematical analysis); Computer science; Video game; Artificial intelligence; Entertainment; Field (mathematics); Deep learning; Domain knowledge; Focus (optics); Hyperparameter; Human–computer interaction; Multimedia","score_opus":0.015724029213907236,"score_gpt":0.2488639597852059,"score_spread":0.23313993057129867,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3167763832","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.100195535,0.0005432973,0.8866397,0.0007243,0.00007712998,0.00008917181,0.00011475379,0.00058310025,0.011033071],"genre_scores_gemma":[0.93046385,0.00018816777,0.061865907,0.0002027066,0.000024422361,0.00009961686,0.00011863281,0.00006088893,0.0069759],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99964,0.00013702473,0.000020259005,0.000075511474,0.000060863644,0.00006635644],"domain_scores_gemma":[0.99919444,0.00042746705,0.00008167998,0.000076275384,0.00015598584,0.00006410883],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011843132,0.0005566982,0.0006470016,0.00021755688,0.0002322174,0.0006661167,0.00081588706,0.0008054352,0.0027494023],"category_scores_gemma":[0.0040347576,0.0003059772,0.0003426041,0.00015327187,0.0008002064,0.0010361508,0.0009791212,0.0016051235,0.00029700014],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005082207,0.00005021221,0.00072817766,0.00004046332,0.000015851287,0.000043828208,0.000038576876,0.96550024,0.00097231986,0.015053762,0.00075071945,0.016755171],"study_design_scores_gemma":[0.000005825669,0.000013989202,0.000078317506,0.0000036143367,0.0000014946681,0.0000046938508,0.0000039628167,0.9942842,0.0002054624,0.0051401868,0.00025624866,0.0000019264498],"about_ca_topic_score_codex":0.0066673206,"about_ca_topic_score_gemma":0.0075674877,"teacher_disagreement_score":0.0066673206,"about_ca_system_score_codex":0.0012172214,"about_ca_system_score_gemma":0.00091547286,"threshold_uncertainty_score":0.013257027},"labels":[],"label_agreement":null},{"id":"W3167916533","doi":"10.48550/arxiv.2106.02617","title":"Be Considerate: Objectives, Side Effects, and Deciding How to Act","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Discretion; Agency (philosophy); Contemplation; Risk analysis (engineering); Work (physics); Computer science; Psychology; Business; Engineering; Political science; Sociology; Law","score_opus":0.054087394248674765,"score_gpt":0.19593051251140975,"score_spread":0.141843118262735,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3167916533","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.47389165,0.0014607701,0.4095353,0.008179621,0.0003099112,0.00026080647,0.00038558154,0.0007935558,0.105182774],"genre_scores_gemma":[0.9675597,0.00028103488,0.026915798,0.00029537547,0.000025140313,0.00009000594,0.00009776744,0.00008941297,0.0046457965],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979576,0.0011278554,0.00008604768,0.0002492823,0.00036787541,0.00021139052],"domain_scores_gemma":[0.9900704,0.006146545,0.0012767555,0.0009090643,0.0007336959,0.0008635731],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031840543,0.0007600323,0.0003758986,0.00036763388,0.00079062616,0.0021620514,0.0006179576,0.0012564373,0.0065310053],"category_scores_gemma":[0.017458746,0.0002962679,0.00040171418,0.00026441854,0.0026066815,0.0029674894,0.0016143281,0.0021721015,0.00067864306],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0029925406,0.0008795922,0.047528565,0.00074133114,0.00027243013,0.0011787948,0.0031151108,0.20029218,0.024339428,0.39070797,0.008476817,0.31947526],"study_design_scores_gemma":[0.00012681016,0.0005589015,0.011582282,0.00015859208,0.00011846505,0.00029037878,0.0010966464,0.29660356,0.010581975,0.66599053,0.012796943,0.000094900555],"about_ca_topic_score_codex":0.00195919,"about_ca_topic_score_gemma":0.003253181,"teacher_disagreement_score":0.0065310053,"about_ca_system_score_codex":0.0009778393,"about_ca_system_score_gemma":0.001266343,"threshold_uncertainty_score":0.02184838},"labels":[],"label_agreement":null},{"id":"W3168124941","doi":"10.48550/arxiv.2106.00922","title":"An Empirical Comparison of Off-policy Prediction Learning Algorithms on the Collision Task","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Backup; Reinforcement learning; Computer science; Algorithm; Bootstrapping (finance); Artificial intelligence; Task (project management); Machine learning; Tree (set theory); Collision; Function (biology); Temporal difference learning; Mathematics; Econometrics","score_opus":0.10285944847924657,"score_gpt":0.26344388981984024,"score_spread":0.1605844413405937,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3168124941","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.88914365,0.0045839865,0.094512455,0.0011253464,0.00030892683,0.00053100474,0.00048762484,0.0021780455,0.0071289577],"genre_scores_gemma":[0.9374929,0.0007288251,0.05802603,0.00029357633,0.000059432034,0.00029597935,0.001216013,0.00027239815,0.0016148116],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9929762,0.0033870733,0.0006580864,0.0012415781,0.0011690023,0.0005680809],"domain_scores_gemma":[0.9164985,0.0691158,0.002075188,0.005480339,0.0052497154,0.0015805028],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015801735,0.0018706325,0.0016804546,0.0017657225,0.0008361588,0.0012183831,0.002342292,0.0031046988,0.0010111189],"category_scores_gemma":[0.060550287,0.00052483415,0.0007826732,0.0012971472,0.0018016221,0.0032316241,0.0020240152,0.0032521512,0.00052433513],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0035930125,0.0027572329,0.026226444,0.0006976983,0.0003999705,0.0001325197,0.00035097133,0.6360968,0.0017935635,0.0033795517,0.0054714577,0.31910086],"study_design_scores_gemma":[0.00020263232,0.001337388,0.005025506,0.00007762918,0.00005571863,0.00009510055,0.00020411176,0.98660046,0.0024648958,0.003014168,0.0008858093,0.00003654047],"about_ca_topic_score_codex":0.00740293,"about_ca_topic_score_gemma":0.006006274,"teacher_disagreement_score":0.015801735,"about_ca_system_score_codex":0.001837911,"about_ca_system_score_gemma":0.002193088,"threshold_uncertainty_score":0.08356857},"labels":[],"label_agreement":null},{"id":"W3168260200","doi":"10.48550/arxiv.2106.11779","title":"Emphatic Algorithms for Deep Reinforcement Learning","year":2021,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Temporal difference learning; Artificial intelligence; Weighting; Convergence (economics); Context (archaeology); Algorithm; Machine learning; Stability (learning theory); Lambda; Function (biology)","score_opus":0.06650257069781514,"score_gpt":0.20297120362475468,"score_spread":0.13646863292693956,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3168260200","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018474272,0.000097621996,0.9963007,0.00012906828,0.000033079672,0.000033250406,0.000019245657,0.00023335723,0.001306336],"genre_scores_gemma":[0.23991267,0.0002798919,0.7538307,0.0004876311,0.000106194064,0.0005212708,0.00017393359,0.00025756733,0.004430077],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990252,0.00040765136,0.00008386968,0.00017005435,0.00025215346,0.00006112068],"domain_scores_gemma":[0.9967494,0.0021273717,0.00022575265,0.0003404751,0.00045474153,0.00010226622],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030587325,0.0009888178,0.00080502126,0.0006026553,0.0004231413,0.0009796867,0.0014940376,0.0014366742,0.005375316],"category_scores_gemma":[0.011691765,0.0006226811,0.00056170113,0.00048545445,0.0013200539,0.0015019159,0.0020382612,0.0032325292,0.0010177948],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013848004,0.00009614781,0.00075528305,0.00019694262,0.000057651425,0.000056827597,0.00010916441,0.6479113,0.0022176434,0.18568861,0.004839935,0.15793203],"study_design_scores_gemma":[0.000018676654,0.000025723904,0.000029530653,0.000011991168,0.000004094649,0.000015274187,0.0000045617194,0.96192795,0.0004971216,0.036309272,0.0011505781,0.000005158065],"about_ca_topic_score_codex":0.0012544343,"about_ca_topic_score_gemma":0.0019006949,"teacher_disagreement_score":0.005375316,"about_ca_system_score_codex":0.0011548377,"about_ca_system_score_gemma":0.00142719,"threshold_uncertainty_score":0.017982244},"labels":[],"label_agreement":null},{"id":"W3169135197","doi":"10.1109/iccma53594.2021.00018","title":"Hyperspace Neighbor Penetration Approach to Dynamic Programming for Model-Based Reinforcement Learning Problems with Slowly Changing Variables in a Continuous State Space","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Telus (Canada)","funders":"","keywords":"Reinforcement learning; Hyperspace; Computer science; Grid; Mathematical optimization; State variable; Computation; Tile; Theoretical computer science; Algorithm; Mathematics; Artificial intelligence; Geometry; Physics","score_opus":0.016641044675257805,"score_gpt":0.2406334646722556,"score_spread":0.2239924199969978,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3169135197","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0043048183,0.00010239687,0.99436426,0.00008058836,0.000019191095,0.000017443053,0.000015571139,0.00008161004,0.0010142649],"genre_scores_gemma":[0.6616724,0.0004287156,0.33200213,0.0003005038,0.00006447262,0.00040932896,0.00015340712,0.0001880598,0.004780983],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99943763,0.00024401958,0.00002150712,0.000103872335,0.0001338278,0.000059141228],"domain_scores_gemma":[0.9989998,0.000672028,0.00007882955,0.000067094734,0.00011874959,0.000063593485],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009484003,0.00075301836,0.0010752011,0.00035803366,0.00033798782,0.0007963861,0.001112043,0.0010467647,0.0032057639],"category_scores_gemma":[0.0022255385,0.00051176327,0.0009110595,0.0004335941,0.0011116989,0.0013314665,0.001629737,0.002318425,0.00033934496],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004826791,0.000025443456,0.00035564555,0.00006222542,0.00003133233,0.000066072,0.00006316322,0.93520206,0.0011561827,0.043351,0.0005259546,0.019112678],"study_design_scores_gemma":[0.0000026413627,0.000010040768,0.000016923812,0.0000022162403,0.0000013254148,0.0000052767464,0.0000025249528,0.994517,0.00012340707,0.0050769118,0.0002396677,0.0000020160091],"about_ca_topic_score_codex":0.003925888,"about_ca_topic_score_gemma":0.0027754835,"teacher_disagreement_score":0.003925888,"about_ca_system_score_codex":0.0008079084,"about_ca_system_score_gemma":0.000851102,"threshold_uncertainty_score":0.010724366},"labels":[],"label_agreement":null},{"id":"W3169292790","doi":"10.48550/arxiv.2106.06854","title":"A Deep Reinforcement Learning Approach to Marginalized Importance\\n Sampling with the Successor Representation","year":2021,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Successor cardinal; Reinforcement learning; Representation (politics); Computer science; Sampling (signal processing); Artificial intelligence; Bridge (graph theory); Variety (cybernetics); Machine learning; Simple random sample; Reinforcement; State (computer science); Mathematics; Algorithm; Sociology; Psychology; Political science; Social psychology; Detector; Law","score_opus":0.09879524359877373,"score_gpt":0.22533058249091992,"score_spread":0.1265353388921462,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3169292790","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013103739,0.00023856867,0.9843551,0.00025076355,0.00005443101,0.00004947019,0.00005243788,0.00039009552,0.0015054146],"genre_scores_gemma":[0.7612417,0.0002381732,0.23302047,0.00029662938,0.00010925985,0.00021483161,0.00020332342,0.00013242463,0.0045431843],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99922407,0.00032405218,0.0000313427,0.0001597794,0.00017120458,0.000089560956],"domain_scores_gemma":[0.9979267,0.0013146664,0.00016560608,0.00021473202,0.00022375559,0.00015453034],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016818463,0.00078543834,0.0014384018,0.0005330995,0.000365677,0.0008762797,0.0019873958,0.0011000418,0.0032133458],"category_scores_gemma":[0.00621963,0.00048020406,0.0005317739,0.0005142713,0.001300317,0.0015342692,0.0016147615,0.0022515818,0.00038667317],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018441233,0.00014639321,0.0015184463,0.00009733123,0.00006473605,0.000088078225,0.00008090379,0.8349187,0.0020201968,0.057018183,0.0028126016,0.10105011],"study_design_scores_gemma":[0.000008751646,0.000020202298,0.000042954205,0.0000047737594,0.0000036301333,0.000008009202,0.0000019986107,0.99056166,0.00025529537,0.008853462,0.00023632431,0.000002937524],"about_ca_topic_score_codex":0.0038193131,"about_ca_topic_score_gemma":0.00473607,"teacher_disagreement_score":0.0038193131,"about_ca_system_score_codex":0.0012907259,"about_ca_system_score_gemma":0.0017551517,"threshold_uncertainty_score":0.010749698},"labels":[],"label_agreement":null},{"id":"W3169514089","doi":"","title":"EMaQ: Expected-Max Q-Learning Operator for Simple Yet Effective Offline and Online RL","year":2021,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Google (Canada); University of Alberta; University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Backup; Heuristic; Operator (biology); Online and offline; Offline learning; Simple (philosophy); Artificial intelligence; Generative grammar; Machine learning; Mathematical optimization; Theoretical computer science; Online learning; Mathematics","score_opus":0.028410596128182336,"score_gpt":0.3196032698915692,"score_spread":0.29119267376338687,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3169514089","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00203276,0.000110071305,0.99495137,0.0002073676,0.00004496626,0.000077403565,0.00004327358,0.0009606149,0.0015722048],"genre_scores_gemma":[0.26727504,0.00025837685,0.7238627,0.0007274397,0.00014811278,0.0006653057,0.0003626652,0.0007105201,0.005989853],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9972796,0.001078184,0.0001622859,0.0005566647,0.0006657904,0.00025750496],"domain_scores_gemma":[0.9943389,0.00382862,0.00029162207,0.00069430034,0.00056693464,0.00027960952],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00519968,0.0017845333,0.0016479384,0.00062826736,0.00061173254,0.0015887809,0.0040137726,0.00255204,0.011852087],"category_scores_gemma":[0.01789861,0.0006401663,0.0008578414,0.0007644854,0.0021209195,0.0024344574,0.004369552,0.004751676,0.0029205687],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043901827,0.0002848199,0.0009984175,0.0003378948,0.000059271933,0.00015362684,0.00023660324,0.5414227,0.0029206432,0.11230149,0.011718981,0.32912657],"study_design_scores_gemma":[0.000030575276,0.00005913452,0.000043630844,0.000015923644,0.0000038879552,0.000028108923,0.000012269157,0.9673505,0.00069554005,0.0302862,0.0014665058,0.000007798892],"about_ca_topic_score_codex":0.0027130109,"about_ca_topic_score_gemma":0.0027706048,"teacher_disagreement_score":0.011852087,"about_ca_system_score_codex":0.0013413841,"about_ca_system_score_gemma":0.002941414,"threshold_uncertainty_score":0.03964919},"labels":[],"label_agreement":null},{"id":"W3170371439","doi":"10.1609/aaai.v36i7.20758","title":"Control-Oriented Model-Based Reinforcement Learning with Implicit Differentiation","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Institute for Advanced Research; Vector Institute; University of Toronto; Université de Montréal","funders":"Compute Canada","keywords":"Reinforcement learning; Bellman equation; Computer science; Context (archaeology); Function (biology); Task (project management); Class (philosophy); Value (mathematics); Likelihood function; Maximum likelihood; Reinforcement; Control (management); Artificial intelligence; Mathematical optimization; Machine learning; Estimation theory; Mathematics; Statistics; Economics; Algorithm","score_opus":0.03202434377017425,"score_gpt":0.25509131036248905,"score_spread":0.2230669665923148,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3170371439","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0151896775,0.0001006874,0.9822948,0.00019348349,0.000020848938,0.000037869748,0.0000150797,0.00026979749,0.0018777142],"genre_scores_gemma":[0.9021251,0.00008327632,0.09509165,0.00015683706,0.000023270655,0.00013693499,0.000039761333,0.00006596928,0.0022773123],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990988,0.00036201434,0.000048332837,0.00015452514,0.00022948967,0.000106842264],"domain_scores_gemma":[0.9969928,0.001892639,0.00035507107,0.00027506857,0.00033336857,0.00015099907],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024287964,0.0011221322,0.0012609,0.00037228988,0.00037673357,0.0013077429,0.0017829619,0.0014450819,0.0022045423],"category_scores_gemma":[0.0070998026,0.0005770535,0.00050743966,0.00037468967,0.0017437439,0.0016546493,0.002072338,0.002477608,0.00039897245],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000056346486,0.00005302559,0.00027345066,0.00004463925,0.000021376478,0.000048313235,0.00005910481,0.9586526,0.0009495644,0.0245665,0.00027739818,0.014997627],"study_design_scores_gemma":[0.000007685009,0.000016560594,0.000012899334,0.0000028810457,0.0000021258802,0.0000043682458,0.000001128098,0.994749,0.00016472899,0.0049646664,0.00007170239,0.0000022219238],"about_ca_topic_score_codex":0.0027318802,"about_ca_topic_score_gemma":0.0024414803,"teacher_disagreement_score":0.0027318802,"about_ca_system_score_codex":0.0012207906,"about_ca_system_score_gemma":0.0015601076,"threshold_uncertainty_score":0.012844861},"labels":[],"label_agreement":null},{"id":"W3171711942","doi":"10.1109/syscon48628.2021.9447080","title":"Optimization of Deep Reinforcement Learning with Hybrid Multi-Task Learning","year":2021,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Task (project management); Machine learning; Deep learning; Learning classifier system; Robot learning; Multi-task learning; Active learning (machine learning); Engineering; Robot","score_opus":0.013230727267571582,"score_gpt":0.2300208503905229,"score_spread":0.21679012312295132,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3171711942","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043623473,0.00031672756,0.9512618,0.0002655221,0.00006674986,0.00007112835,0.000033244418,0.00039088336,0.0039704223],"genre_scores_gemma":[0.94054776,0.00009574057,0.056500334,0.00012177267,0.000027602857,0.00014781403,0.000046833426,0.000034299213,0.0024778864],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99948406,0.00016379743,0.00002422574,0.000108125634,0.00011980499,0.0000998935],"domain_scores_gemma":[0.9991781,0.00041809352,0.00011645044,0.000053718402,0.00016236193,0.00007125218],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014004323,0.00094929256,0.00090754626,0.00033117883,0.00027902544,0.0007743155,0.0011420246,0.0011072013,0.0015711839],"category_scores_gemma":[0.002197598,0.00038305545,0.00046022297,0.0002764611,0.0008716422,0.0007145113,0.0013274419,0.0010131294,0.00020650771],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000045777477,0.000035615445,0.00021302712,0.000022330234,0.00002389191,0.000042919473,0.000016867822,0.98573095,0.00078205945,0.0021966307,0.00024609626,0.010643836],"study_design_scores_gemma":[0.0000056630515,0.000016657812,0.0000214246,0.0000013051291,0.0000019407312,0.0000033045774,0.0000013543306,0.9991184,0.000118012205,0.000635116,0.00007560204,0.0000012379655],"about_ca_topic_score_codex":0.0036864008,"about_ca_topic_score_gemma":0.0027768195,"teacher_disagreement_score":0.0036864008,"about_ca_system_score_codex":0.00081056496,"about_ca_system_score_gemma":0.001276367,"threshold_uncertainty_score":0.0074062347},"labels":[],"label_agreement":null},{"id":"W3172545192","doi":"10.1007/978-3-031-26889-2_19","title":"Least-Restrictive Multi-agent Collision Avoidance via Deep Meta Reinforcement Learning and Optimal Control","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in networks and systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Regina; Simon Fraser University","funders":"","keywords":"Reinforcement learning; Collision avoidance; Computer science; Collision; Function (biology); Interrupt; Trajectory; Artificial intelligence; Autonomous agent; Control (management); Control theory (sociology); Computer security","score_opus":0.023811569614677456,"score_gpt":0.24225485058398818,"score_spread":0.21844328096931073,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3172545192","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017662484,0.0002798762,0.97648084,0.00020923132,0.00005359385,0.000021365908,0.000029391047,0.00031851162,0.0049447655],"genre_scores_gemma":[0.9114031,0.00016417366,0.08242605,0.00015034263,0.00004810402,0.000109834735,0.00006322322,0.00010262181,0.0055325637],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996549,0.000083271996,0.000017395729,0.000078044905,0.000089771675,0.00007660687],"domain_scores_gemma":[0.99909985,0.00050432613,0.000113875736,0.00009969068,0.000116308816,0.000065957465],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007930282,0.0010142298,0.0014223488,0.00038683816,0.0005377031,0.0010165302,0.0018487031,0.001503265,0.002397016],"category_scores_gemma":[0.002202411,0.0007958178,0.0007591235,0.00040612725,0.0012855221,0.0011953311,0.002893964,0.0021692468,0.00040057348],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000050565108,0.000024829475,0.0001338383,0.000030585812,0.000030619667,0.00002911188,0.000028563267,0.9734364,0.0010391006,0.009963139,0.00060883875,0.01462447],"study_design_scores_gemma":[0.0000028255267,0.000009768239,0.000018540324,0.000002129795,0.0000022205497,0.0000038952835,0.0000019255301,0.9970458,0.0000934819,0.0027505497,0.000067044464,0.000001796384],"about_ca_topic_score_codex":0.004752935,"about_ca_topic_score_gemma":0.0042717378,"teacher_disagreement_score":0.004752935,"about_ca_system_score_codex":0.0008867347,"about_ca_system_score_gemma":0.0009271195,"threshold_uncertainty_score":0.009450555},"labels":[],"label_agreement":null},{"id":"W3173549437","doi":"10.1609/aaai.v35i13.17378","title":"How RL Agents Behave When Their Actions Are Modified","year":2021,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Action (physics); Supervisor; Markov decision process; Computer science; Process (computing); Risk analysis (engineering); Reinforcement; Control (management); Intervention (counseling); Artificial intelligence; Q-learning; Machine learning; Markov process; Psychology; Social psychology; Business; Economics; Mathematics","score_opus":0.0900760941624977,"score_gpt":0.27071665494807196,"score_spread":0.18064056078557425,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3173549437","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3442026,0.0005377508,0.63174134,0.00343971,0.00016275901,0.0001158385,0.00014578816,0.0015062463,0.018147882],"genre_scores_gemma":[0.97231174,0.00019550168,0.024731116,0.00019398323,0.000016053495,0.00004567671,0.000042980842,0.00010420433,0.002358739],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99895716,0.00051322614,0.000045067674,0.00018054305,0.00017568663,0.00012845272],"domain_scores_gemma":[0.99751866,0.0011749538,0.00040350138,0.00044632595,0.00029812622,0.00015850678],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001581463,0.00036336257,0.00034343667,0.0002494132,0.0002852192,0.0012313635,0.00073833886,0.001070848,0.0013060842],"category_scores_gemma":[0.009810132,0.0003142186,0.00029122844,0.0001446363,0.0014879806,0.0020897912,0.000613134,0.0010139076,0.00052713335],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017805683,0.00008799494,0.00901522,0.0001187101,0.00013720982,0.00029901176,0.0006037204,0.8582607,0.011520282,0.06899774,0.0022061765,0.048575282],"study_design_scores_gemma":[0.000032779997,0.000049398943,0.0014427183,0.000026479074,0.000022050097,0.00007475511,0.00011266175,0.93755436,0.0024494138,0.05658626,0.0016194362,0.00002972135],"about_ca_topic_score_codex":0.0040488928,"about_ca_topic_score_gemma":0.002545191,"teacher_disagreement_score":0.0040488928,"about_ca_system_score_codex":0.0007888103,"about_ca_system_score_gemma":0.00075372047,"threshold_uncertainty_score":0.008363664},"labels":[],"label_agreement":null},{"id":"W3175558129","doi":"10.1609/aaai.v35i12.17276","title":"Improving Sample Efficiency in Model-Free Reinforcement Learning from Images","year":2021,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":201,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Robustness (evolution); Computer science; Artificial intelligence; Stability (learning theory); Machine learning; Encoder; Representation (politics); Noise (video); Code (set theory); Image (mathematics)","score_opus":0.019185411067575067,"score_gpt":0.23646707892105248,"score_spread":0.2172816678534774,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3175558129","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038278405,0.00018226294,0.9582589,0.00030321366,0.000025856052,0.000053842352,0.00003297544,0.0010955192,0.0017689874],"genre_scores_gemma":[0.8619704,0.00009982718,0.13523075,0.00021187334,0.00002748986,0.00015168644,0.00011029416,0.00024015426,0.0019574508],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99920064,0.00032844924,0.00003865067,0.00017381103,0.00016012313,0.00009839044],"domain_scores_gemma":[0.9949868,0.0035899798,0.00034713012,0.00058974116,0.00030962942,0.0001767602],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028771795,0.0010893969,0.0014901061,0.0004586175,0.0005317111,0.0009690191,0.00191144,0.001345246,0.0025873862],"category_scores_gemma":[0.013777561,0.0007505636,0.0005369825,0.00031585037,0.0017557859,0.0020839553,0.0019106969,0.0022466592,0.0005209126],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023454872,0.000134606,0.00107486,0.00009412781,0.000040328036,0.000077081444,0.00010547979,0.9261957,0.0025013017,0.01738233,0.0009957497,0.051163938],"study_design_scores_gemma":[0.000013754284,0.000024736422,0.00004657449,0.0000053403387,0.00000283055,0.000007565566,0.0000035670569,0.9945098,0.000486509,0.004798462,0.00009816155,0.000002773098],"about_ca_topic_score_codex":0.0048034,"about_ca_topic_score_gemma":0.005092191,"teacher_disagreement_score":0.0048034,"about_ca_system_score_codex":0.0011778348,"about_ca_system_score_gemma":0.0016386191,"threshold_uncertainty_score":0.015216112},"labels":[],"label_agreement":null},{"id":"W3176022961","doi":"10.1145/3449639.3459348","title":"On the impact of tangled program graph marking schemes under the atari reinforcement learning benchmark","year":2021,"lang":"en","type":"article","venue":"Proceedings of the Genetic and Evolutionary Computation Conference","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Reinforcement learning; Benchmark (surveying); Graph; Adaptation (eye); Heuristic; Scheme (mathematics); Theoretical computer science; Modularity (biology); Artificial intelligence; Node (physics); Machine learning; Engineering; Mathematics","score_opus":0.0203845929356071,"score_gpt":0.25870910566226896,"score_spread":0.23832451272666186,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3176022961","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98292875,0.00062890537,0.006829748,0.0005195426,0.00006277929,0.000055181692,0.00030586863,0.0006843286,0.007984968],"genre_scores_gemma":[0.991602,0.00009209328,0.0070392997,0.000081417646,0.000010535166,0.00003074034,0.0003466074,0.00007659977,0.0007207352],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998123,0.0008871703,0.00011084009,0.00025163905,0.00029826048,0.00032900114],"domain_scores_gemma":[0.9673177,0.025965033,0.0015656516,0.0020834992,0.0015484898,0.0015196837],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004013534,0.0009272197,0.0005331166,0.0009376683,0.00059320754,0.0009867746,0.0013145505,0.0010325693,0.002426487],"category_scores_gemma":[0.023439683,0.00021550924,0.00030560288,0.00063859194,0.000999537,0.0015322419,0.0010823858,0.0014573323,0.00026878534],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018615164,0.0012630954,0.011825692,0.00029077276,0.0001182599,0.000129994,0.00010870286,0.9114919,0.00455171,0.006124481,0.0034353866,0.05879855],"study_design_scores_gemma":[0.00018097175,0.0013973017,0.0041234572,0.00004956938,0.000054428234,0.000043071224,0.00014057421,0.9823889,0.0045228833,0.0062085683,0.0008694576,0.000020771728],"about_ca_topic_score_codex":0.007742818,"about_ca_topic_score_gemma":0.011996714,"teacher_disagreement_score":0.007742818,"about_ca_system_score_codex":0.0014147979,"about_ca_system_score_gemma":0.001211096,"threshold_uncertainty_score":0.02122587},"labels":[],"label_agreement":null},{"id":"W3176342773","doi":"","title":"OptiDICE: Offline Policy Optimization via Stationary Distribution Correction Estimation","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Hyperparameter; Reinforcement learning; Benchmark (surveying); Computer science; Bootstrapping (finance); Set (abstract data type); Mathematical optimization; Online and offline; Offline learning; Artificial intelligence; Machine learning; Mathematics; Online learning; Econometrics","score_opus":0.034880791229417715,"score_gpt":0.20066294963608636,"score_spread":0.16578215840666866,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3176342773","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009484437,0.000466906,0.9818772,0.00034321228,0.00015046325,0.00013233608,0.00013632195,0.0039838143,0.0034253495],"genre_scores_gemma":[0.44395003,0.00043027886,0.54472584,0.00078497035,0.00016940966,0.0004527809,0.0008356369,0.0010643685,0.007586744],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990023,0.00027367737,0.000060972565,0.0003127572,0.00022916112,0.00012122328],"domain_scores_gemma":[0.9967577,0.0021057553,0.00023384062,0.00047020311,0.00028488203,0.00014755191],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017985684,0.0016788107,0.0022062631,0.000717259,0.00052517117,0.0017160102,0.002236865,0.0018629592,0.0038599996],"category_scores_gemma":[0.0096214395,0.0008725745,0.0007147885,0.00069741934,0.0012331782,0.0019195682,0.0021553277,0.0035886543,0.0013274723],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018269075,0.00020339114,0.000984475,0.0001347996,0.000073348005,0.000078167875,0.000066834706,0.80173063,0.0014190725,0.008775121,0.0063853823,0.17996612],"study_design_scores_gemma":[0.000028530641,0.000030369176,0.000064130436,0.000010033265,0.000004547257,0.000018272152,0.0000056471795,0.99441105,0.0007021969,0.0036806893,0.0010374191,0.0000070415936],"about_ca_topic_score_codex":0.007563427,"about_ca_topic_score_gemma":0.0077425097,"teacher_disagreement_score":0.007563427,"about_ca_system_score_codex":0.0012093794,"about_ca_system_score_gemma":0.003723854,"threshold_uncertainty_score":0.015038788},"labels":[],"label_agreement":null},{"id":"W3176644733","doi":"10.1016/j.jprocont.2021.06.004","title":"Online reinforcement learning for a continuous space system with experimental validation","year":2021,"lang":"en","type":"article","venue":"Journal of Process Control","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; State space; Artificial intelligence; Machine learning; Heuristic; Trajectory; Asynchronous communication; Mathematics","score_opus":0.012318515279714703,"score_gpt":0.2662024340434771,"score_spread":0.2538839187637624,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3176644733","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5252957,0.00028330993,0.46838427,0.0004658852,0.00012289667,0.00030762973,0.00022641267,0.0009724162,0.003941572],"genre_scores_gemma":[0.98866314,0.00001483326,0.010620906,0.000014776065,0.000002767387,0.00007356402,0.000046952642,0.000012858903,0.00055019226],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99893254,0.00044946145,0.000072549854,0.00016413507,0.00024989643,0.00013137807],"domain_scores_gemma":[0.9877286,0.0082875425,0.0008822093,0.0009422926,0.0019139482,0.00024548842],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004842431,0.00080330507,0.0007245813,0.00045585592,0.00056995486,0.0008000032,0.001171312,0.0013836359,0.0029066245],"category_scores_gemma":[0.013081998,0.0003452593,0.00049011706,0.00021755605,0.0015526467,0.0008890641,0.001415336,0.0012871486,0.00031256606],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001016194,0.00029568345,0.0015375308,0.00027215955,0.00006545417,0.00012628792,0.0001592234,0.9661995,0.009084319,0.0036401248,0.00033089236,0.017272662],"study_design_scores_gemma":[0.00006150654,0.00021108653,0.00045550163,0.000009689777,0.000009290536,0.000013824704,0.000009104343,0.99514353,0.0033202195,0.00064643746,0.000108442604,0.0000114016675],"about_ca_topic_score_codex":0.009307378,"about_ca_topic_score_gemma":0.0048832474,"teacher_disagreement_score":0.009307378,"about_ca_system_score_codex":0.0010324199,"about_ca_system_score_gemma":0.0014221199,"threshold_uncertainty_score":0.025609553},"labels":[],"label_agreement":null},{"id":"W3176880374","doi":"10.1109/lra.2021.3093551","title":"A Sim-to-Real Pipeline for Deep Reinforcement Learning for Autonomous Robot Navigation in Cluttered Rough Terrain","year":2021,"lang":"en","type":"article","venue":"IEEE Robotics and Automation Letters","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":84,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Traverse; Terrain; Reinforcement learning; Robot; Pipeline (software); Artificial intelligence; Computer science; Computer vision; Point (geometry); Mobile robot; Trajectory; Geography; Mathematics; Cartography","score_opus":0.01694845159004387,"score_gpt":0.26733563832348006,"score_spread":0.2503871867334362,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3176880374","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020086264,0.00010636973,0.973277,0.0001552719,0.000039223858,0.00007821572,0.000056676494,0.003783225,0.0024178098],"genre_scores_gemma":[0.5833906,0.0000877495,0.4118055,0.0001982615,0.000015110171,0.00015359391,0.00017775623,0.00016541389,0.004005944],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998379,0.000027087772,0.000008645693,0.00004768001,0.000049076712,0.00002952824],"domain_scores_gemma":[0.9997501,0.00007083638,0.000026755753,0.000048273796,0.00006444303,0.000039627303],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00048434883,0.0005867925,0.0004075343,0.00021152422,0.00024953997,0.0003355993,0.0012059027,0.0006287972,0.0036168261],"category_scores_gemma":[0.0010294615,0.0003467959,0.00030480037,0.00014046686,0.0005470909,0.00077696494,0.0010796459,0.0011797952,0.00070386374],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002487169,0.00027794496,0.0018238076,0.00015256084,0.000052249445,0.00025338356,0.00015483983,0.6349163,0.031658437,0.0125400685,0.0058857286,0.31203598],"study_design_scores_gemma":[0.000008783651,0.000053595537,0.00007575306,0.0000032190885,0.0000024658223,0.000018990153,0.0000034886536,0.99559206,0.0024344062,0.00089187187,0.00091105845,0.0000042766223],"about_ca_topic_score_codex":0.004151918,"about_ca_topic_score_gemma":0.005426053,"teacher_disagreement_score":0.004151918,"about_ca_system_score_codex":0.00064015255,"about_ca_system_score_gemma":0.0011624619,"threshold_uncertainty_score":0.0120995045},"labels":[],"label_agreement":null},{"id":"W3177370677","doi":"10.1609/aaai.v35i18.17927","title":"Solving JumpIN’ Using Zero-Dependency Reinforcement Learning (Student Abstract)","year":2021,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Reinforcement learning; Computer science; Dependency grammar; Python (programming language); Interpretability; Dependency (UML); GRASP; Modular design; Backtracking; Artificial intelligence; Algorithm; Programming language","score_opus":0.08900379891208272,"score_gpt":0.31973577856656876,"score_spread":0.23073197965448605,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3177370677","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.083978236,0.00004489669,0.89861715,0.00055446615,0.00008374641,0.0001521995,0.00014003154,0.005161967,0.011267314],"genre_scores_gemma":[0.75485533,0.00003053054,0.23882967,0.00024435742,0.000018102603,0.00019869833,0.00020036753,0.000342552,0.005280323],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995264,0.00015073695,0.00002379836,0.00009524592,0.00011364688,0.00009010525],"domain_scores_gemma":[0.9986726,0.0008368267,0.00009559727,0.00016581727,0.00012398064,0.000105297695],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012312967,0.0004995131,0.00049861835,0.00021385003,0.00034101517,0.00062768615,0.0014159429,0.0007629989,0.009138077],"category_scores_gemma":[0.004190159,0.0002826594,0.0005435396,0.00016121802,0.0011604445,0.0010450607,0.0016061073,0.0013831042,0.0007659168],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049646624,0.0005175594,0.004018009,0.00023824483,0.000074755735,0.00025272518,0.00024789872,0.7808319,0.0077554705,0.06729143,0.007868478,0.13040714],"study_design_scores_gemma":[0.000028055094,0.00003289665,0.00012306955,0.0000049482173,0.0000030601364,0.000011366383,0.000006436433,0.98621714,0.0020900615,0.010748271,0.0007300874,0.0000045392594],"about_ca_topic_score_codex":0.0038585074,"about_ca_topic_score_gemma":0.0045918645,"teacher_disagreement_score":0.009138077,"about_ca_system_score_codex":0.00068585336,"about_ca_system_score_gemma":0.001234799,"threshold_uncertainty_score":0.030569911},"labels":[],"label_agreement":null},{"id":"W3177464508","doi":"","title":"On the Sample Complexity of Batch Reinforcement Learning with Policy-Induced Data.","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Omega; Sample complexity; Horizon; Markov decision process; Upper and lower bounds; Time horizon; Function (biology); Exponential function; Combinatorics; Mathematics; Sample (material); Finite set; Binary logarithm; Markov process; Mathematical optimization; Physics; Computer science; Statistics; Artificial intelligence; Mathematical analysis; Quantum mechanics","score_opus":0.20235586642382825,"score_gpt":0.2316189306737228,"score_spread":0.029263064249894555,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3177464508","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.091667674,0.002896925,0.891506,0.0045617507,0.00020681729,0.00038636863,0.00086621934,0.0009205999,0.0069875964],"genre_scores_gemma":[0.81271964,0.0017845233,0.17580211,0.001366898,0.00051574013,0.0011815457,0.0017272538,0.00055803056,0.004344291],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9943037,0.002547208,0.00031591908,0.0011082228,0.0010810003,0.0006439975],"domain_scores_gemma":[0.85505706,0.13327627,0.0035660225,0.0044953995,0.0020055303,0.001599735],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011007392,0.0020988928,0.003125587,0.0013245303,0.0011755017,0.0030092988,0.0036336624,0.003295624,0.0059654503],"category_scores_gemma":[0.07236342,0.0012879592,0.0019914925,0.0011792431,0.0041959053,0.008292516,0.0040289424,0.007149305,0.00055057724],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000932378,0.00036938392,0.0034594159,0.00044527336,0.00024328985,0.00013191272,0.00015080746,0.8774789,0.0013241336,0.09177452,0.002400396,0.021289583],"study_design_scores_gemma":[0.000040937583,0.00007093714,0.00023761157,0.000025899726,0.000017471182,0.000019009605,0.000013831431,0.944951,0.000358109,0.0540437,0.00020883972,0.0000126505975],"about_ca_topic_score_codex":0.005583488,"about_ca_topic_score_gemma":0.005626652,"teacher_disagreement_score":0.011007392,"about_ca_system_score_codex":0.005132472,"about_ca_system_score_gemma":0.004838794,"threshold_uncertainty_score":0.058213353},"labels":[],"label_agreement":null},{"id":"W3181012797","doi":"10.1609/aaai.v36i6.20660","title":"Learning Expected Emphatic Traces for Deep RL","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Weighting; Scalability; Convergence (economics); Stability (learning theory); Artificial intelligence; Machine learning; Sampling (signal processing); Sample (material); Artificial neural network; Baseline (sea); Key (lock); Variance (accounting); Algorithm; Computer vision","score_opus":0.0655002341992619,"score_gpt":0.2902482295044936,"score_spread":0.22474799530523168,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3181012797","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017422672,0.00008106986,0.98075575,0.00015527038,0.000023136465,0.00003076936,0.000045621113,0.0006974584,0.0007882821],"genre_scores_gemma":[0.80523545,0.00012708569,0.19023274,0.00018003,0.000041106767,0.0002207219,0.00024925347,0.0002416313,0.003472049],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994128,0.00019409033,0.000046568308,0.00013037371,0.00014491846,0.000071269256],"domain_scores_gemma":[0.9959472,0.0026691856,0.0003020169,0.00041096707,0.00046645055,0.00020416103],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019976052,0.0010371582,0.00092167425,0.0006188543,0.00038452717,0.001056403,0.0017566192,0.0009589416,0.0025197698],"category_scores_gemma":[0.014162529,0.0007893209,0.00044504358,0.00043997524,0.0012043167,0.002203086,0.0019219276,0.0023268922,0.0004432451],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016515538,0.00006257475,0.0009740539,0.00006988085,0.000033119835,0.00006179462,0.00010075701,0.9214735,0.0017543158,0.01815574,0.0007895121,0.056359705],"study_design_scores_gemma":[0.000005146296,0.000010983843,0.000027751237,0.0000030754709,0.0000014262631,0.0000036052381,0.0000032306486,0.99348336,0.00029969422,0.006070542,0.00008890963,0.0000022683914],"about_ca_topic_score_codex":0.0033657022,"about_ca_topic_score_gemma":0.0047495323,"teacher_disagreement_score":0.0033657022,"about_ca_system_score_codex":0.001159165,"about_ca_system_score_gemma":0.0014515277,"threshold_uncertainty_score":0.010564446},"labels":[],"label_agreement":null},{"id":"W3181841583","doi":"10.1109/crv52889.2021.00018","title":"Uncertainty-Aware Policy Sampling and Mixing for Safe Interactive Imitation Learning","year":2021,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Université de Montréal","funders":"","keywords":"Computer science; Trajectory; Process (computing); Imitation; Interactivity; Harm; Robot; Benchmark (surveying); Mixing (physics); Interleaving; Sampling (signal processing); Function (biology); Artificial intelligence; Human–computer interaction; Machine learning; Multimedia; Computer vision; Programming language","score_opus":0.03195446967103175,"score_gpt":0.31911333008798,"score_spread":0.2871588604169482,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3181841583","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014388826,0.00008031219,0.9842009,0.00012418204,0.000013767038,0.00005471104,0.000013839478,0.00036593413,0.00075760134],"genre_scores_gemma":[0.8345017,0.00009916701,0.16319399,0.00013360412,0.000042598596,0.0002682222,0.000067570196,0.00011803797,0.0015750515],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977664,0.0009099518,0.00012899819,0.00036506864,0.0006435262,0.00018607796],"domain_scores_gemma":[0.9915081,0.00644048,0.00062482746,0.00069179706,0.0003735273,0.0003613128],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031659533,0.0010457644,0.0011947715,0.00059506245,0.00065182557,0.0010030513,0.0016792606,0.0015212529,0.0019694318],"category_scores_gemma":[0.017287696,0.0007207202,0.00078323315,0.00033613562,0.0022884065,0.0019530803,0.0033439475,0.002744877,0.0003668958],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045789345,0.00018908137,0.0017367828,0.000121885365,0.00006503343,0.0001804573,0.0004561035,0.87104243,0.0062801586,0.041409343,0.0005525077,0.077508375],"study_design_scores_gemma":[0.000019940728,0.00006259557,0.00008610867,0.000008784959,0.0000058874452,0.000021834454,0.000011377291,0.98581463,0.0017374798,0.011932531,0.00029009313,0.000008744411],"about_ca_topic_score_codex":0.0019248467,"about_ca_topic_score_gemma":0.0016859198,"teacher_disagreement_score":0.0031659533,"about_ca_system_score_codex":0.0010737972,"about_ca_system_score_gemma":0.0015699929,"threshold_uncertainty_score":0.016743422},"labels":[],"label_agreement":null},{"id":"W3184031856","doi":"10.22215/etd/2021-14448","title":"Several Reinforcement Learning Methods in Mean-Field Games with Binary Action Spaces","year":2021,"lang":"en","type":"dissertation","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Reinforcement learning; Action (physics); Convergence (economics); Computer science; Binary number; Task (project management); Field (mathematics); Artificial intelligence; Population; Binary classification; Machine learning; Space (punctuation); Mathematical optimization; Mathematics; Engineering; Support vector machine","score_opus":0.027511521246958562,"score_gpt":0.3446213576646282,"score_spread":0.31710983641766965,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3184031856","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010813921,0.0027222876,0.9772504,0.00088700943,0.00019370695,0.00009920245,0.00004598434,0.000104354454,0.007883099],"genre_scores_gemma":[0.5410171,0.0045544086,0.43223867,0.000713071,0.00042319956,0.00067381497,0.00017808184,0.00016442273,0.020037195],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991812,0.00038968527,0.000041730982,0.00013971858,0.00017436032,0.00007329249],"domain_scores_gemma":[0.9977673,0.0016571415,0.00012368103,0.0000662382,0.00026077023,0.00012491028],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024928723,0.0010861901,0.0013460085,0.00093775365,0.00070353353,0.0011412427,0.0016487562,0.0018829699,0.003663888],"category_scores_gemma":[0.0060642147,0.00048188554,0.0012885365,0.0007521778,0.0014493605,0.0014315315,0.0011816651,0.0026157596,0.0004333065],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000075143034,0.00012365115,0.00060521177,0.00024382847,0.00008606782,0.000039962473,0.00013176649,0.7856351,0.00063542766,0.13002902,0.0022340473,0.08016076],"study_design_scores_gemma":[0.00001896294,0.000026651678,0.00006943947,0.000020627369,0.0000103619395,0.000010757063,0.000007829401,0.97770566,0.00013567715,0.020941468,0.0010440053,0.000008625755],"about_ca_topic_score_codex":0.0048316154,"about_ca_topic_score_gemma":0.003500876,"teacher_disagreement_score":0.0048316154,"about_ca_system_score_codex":0.0018682986,"about_ca_system_score_gemma":0.001397082,"threshold_uncertainty_score":0.013555467},"labels":[],"label_agreement":null},{"id":"W3185313117","doi":"","title":"Exploration-Driven Representation Learning in Reinforcement Learning","year":2021,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Representation (politics); Machine learning","score_opus":0.06580708750621926,"score_gpt":0.32819784165969396,"score_spread":0.2623907541534747,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3185313117","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016902657,0.000500291,0.9800181,0.0004054369,0.000065737084,0.000033548844,0.00003128483,0.00018423327,0.0018586735],"genre_scores_gemma":[0.8632681,0.00041321432,0.13049988,0.00024754243,0.00008139807,0.00024085023,0.00011407216,0.00012173064,0.0050132363],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990627,0.00054911827,0.000041002848,0.00012757917,0.00013305082,0.000086599466],"domain_scores_gemma":[0.9951717,0.0039010672,0.00017411455,0.0002197694,0.00038914467,0.00014414532],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028760538,0.00059396366,0.001413371,0.00047037244,0.00039658524,0.0010638823,0.0016991696,0.0016170694,0.003295768],"category_scores_gemma":[0.009238759,0.00067117193,0.0005677511,0.00061796437,0.0015631744,0.0020809518,0.0021377993,0.0023055968,0.0003331587],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013284014,0.00008668648,0.0005919084,0.00010546299,0.000054642245,0.00004582067,0.0000826308,0.8887298,0.0008470824,0.05299797,0.0012833009,0.05504181],"study_design_scores_gemma":[0.000009212011,0.000018489422,0.000022406559,0.000004238905,0.0000028872382,0.0000042262996,0.000002590736,0.9889577,0.00011507874,0.0107265385,0.00013414233,0.0000025051818],"about_ca_topic_score_codex":0.0036428268,"about_ca_topic_score_gemma":0.002481311,"teacher_disagreement_score":0.0036428268,"about_ca_system_score_codex":0.0011068643,"about_ca_system_score_gemma":0.0012429482,"threshold_uncertainty_score":0.015210211},"labels":[],"label_agreement":null},{"id":"W3185541071","doi":"","title":"Contextual Policy Transfer in Reinforcement Learning Domains via Deep Mixtures-of-Experts","year":2021,"lang":"en","type":"article","venue":"Uncertainty in Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Robustness (evolution); Machine learning; Artificial intelligence; Transfer of learning; Reuse; Context (archaeology); Task (project management); Bayesian probability; Engineering","score_opus":0.03181191723208893,"score_gpt":0.2972237819968313,"score_spread":0.2654118647647424,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3185541071","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049283657,0.0005257522,0.9467129,0.00032517564,0.000051359984,0.00008450114,0.000059103248,0.0011314718,0.0018261482],"genre_scores_gemma":[0.89965385,0.00018314556,0.09770652,0.00023767319,0.000039733914,0.0001477966,0.00011470098,0.00011249329,0.0018040608],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986381,0.00059898506,0.000064091655,0.00029651797,0.00022719715,0.00017518285],"domain_scores_gemma":[0.99600005,0.002854253,0.0002896045,0.00030199578,0.00031748795,0.00023660016],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035245605,0.001447765,0.0017417886,0.0006059097,0.0004881209,0.0010266951,0.002157267,0.0018580519,0.0021478473],"category_scores_gemma":[0.010889275,0.00088425033,0.0008115894,0.00052260427,0.001693613,0.0021679353,0.002774528,0.0031755855,0.00044007465],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015351675,0.00009070586,0.0005535864,0.000049423357,0.000042847558,0.000047103735,0.000100465724,0.9580431,0.0007348411,0.006849368,0.00052537164,0.03280957],"study_design_scores_gemma":[0.00001191632,0.000024089426,0.000040343773,0.0000048328734,0.0000040403315,0.000004863537,0.0000046084997,0.99402606,0.00029880315,0.0054463395,0.00012949306,0.0000045979723],"about_ca_topic_score_codex":0.007612758,"about_ca_topic_score_gemma":0.0061775222,"teacher_disagreement_score":0.007612758,"about_ca_system_score_codex":0.0017019169,"about_ca_system_score_gemma":0.0016000636,"threshold_uncertainty_score":0.018639863},"labels":[],"label_agreement":null},{"id":"W3185619002","doi":"10.48550/arxiv.2107.08114","title":"Decentralized Multi-Agent Reinforcement Learning for Task Offloading Under Uncertainty","year":2021,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Marl; Computer science; Robustness (evolution); Task (project management); Curse of dimensionality; Artificial intelligence; Machine learning; Distributed computing; Engineering","score_opus":0.0788413178322847,"score_gpt":0.3097153766254181,"score_spread":0.23087405879313339,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3185619002","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06269987,0.00042041094,0.93199486,0.00058372953,0.00008928277,0.00006710672,0.000059816208,0.00053262286,0.0035523279],"genre_scores_gemma":[0.97028726,0.0001022036,0.027758157,0.00011893966,0.000028872859,0.0000868099,0.000044065087,0.000036737438,0.0015368894],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994905,0.00017689755,0.000023219784,0.00010959523,0.00009170211,0.00010796539],"domain_scores_gemma":[0.9978963,0.0012989668,0.00027314897,0.00014616623,0.00019988426,0.00018560112],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012310704,0.0007791763,0.0010523349,0.00025592835,0.0004047081,0.0006820529,0.0010759906,0.0008338606,0.0018019757],"category_scores_gemma":[0.0047755144,0.00041372428,0.00031826706,0.00025723345,0.0011090422,0.00091327267,0.0013494394,0.0015698577,0.0002466967],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000099809964,0.000059887436,0.0005005718,0.000050451446,0.000027524256,0.000055280332,0.000038048405,0.9732046,0.0011169955,0.0071032313,0.0007978111,0.016945727],"study_design_scores_gemma":[0.000010035355,0.0000152349185,0.00004416698,0.00000238026,0.0000023733376,0.000004499617,0.0000032740875,0.99583215,0.00012855648,0.0038293537,0.0001259695,0.0000019716738],"about_ca_topic_score_codex":0.0032390882,"about_ca_topic_score_gemma":0.0031734474,"teacher_disagreement_score":0.0032390882,"about_ca_system_score_codex":0.0008341256,"about_ca_system_score_gemma":0.0012882015,"threshold_uncertainty_score":0.0065106153},"labels":[],"label_agreement":null},{"id":"W3187079608","doi":"10.24963/ijcai.2021/562","title":"Symbolic Dynamic Programming for Continuous State MDPs with Linear Program Transitions","year":2021,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Toronto","funders":"","keywords":"Discretization; Computer science; Piecewise linear function; Dynamic programming; Linear programming; Mathematical optimization; Operator (biology); Bellman equation; Piecewise; State (computer science); Algorithm; Mathematics","score_opus":0.010607059097889362,"score_gpt":0.2697420779943324,"score_spread":0.2591350188964431,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3187079608","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007055212,0.00012136708,0.9895176,0.0001940086,0.000017534481,0.000041026073,0.0000583261,0.00018420852,0.002810742],"genre_scores_gemma":[0.5682869,0.00038309602,0.4254877,0.00013967506,0.000037841626,0.0005681028,0.000277705,0.00016431476,0.00465467],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99940515,0.00025451914,0.000027666323,0.00010512661,0.00014831254,0.000059298905],"domain_scores_gemma":[0.9972042,0.0023167739,0.00016141144,0.000087768094,0.00014918412,0.00008072686],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011869842,0.00091012585,0.00083887944,0.00044002093,0.0004378112,0.0010927452,0.0007230962,0.00092535885,0.0038975503],"category_scores_gemma":[0.0046504326,0.00043362507,0.0007793189,0.00052885857,0.0016915125,0.00079824677,0.0015515811,0.0019849695,0.00033029314],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000015586216,0.000012848081,0.00011434641,0.000046560166,0.000008534321,0.000036109188,0.00004291455,0.96046674,0.00026952368,0.030947207,0.0002578547,0.0077818013],"study_design_scores_gemma":[0.000007598655,0.000006555168,0.000011219604,0.000006379771,0.0000015914907,0.000004629424,0.0000079029105,0.9820071,0.00013729483,0.017433401,0.0003742403,0.0000020452733],"about_ca_topic_score_codex":0.0046148524,"about_ca_topic_score_gemma":0.004538363,"teacher_disagreement_score":0.0046148524,"about_ca_system_score_codex":0.001306595,"about_ca_system_score_gemma":0.00195929,"threshold_uncertainty_score":0.013038576},"labels":[],"label_agreement":null},{"id":"W3190031629","doi":"10.48550/arxiv.2108.02827","title":"An Elementary Proof that Q-learning Converges Almost Surely","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Q-learning; Function (biology); Bellman equation; Field (mathematics); Computer science; State (computer science); Value (mathematics); Elementary proof; Mathematical economics; Artificial intelligence; Connection (principal bundle); Mathematics; Discrete mathematics; Algorithm; Machine learning; Pure mathematics","score_opus":0.07020033359668047,"score_gpt":0.19847805084323028,"score_spread":0.1282777172465498,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3190031629","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00593015,0.0011771404,0.9544432,0.007041969,0.0005274266,0.00013070396,0.00045741146,0.0003481977,0.029943813],"genre_scores_gemma":[0.51464206,0.006420852,0.42082155,0.012769437,0.0023894305,0.0020486198,0.0010165235,0.0009407551,0.038950857],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99653363,0.0010085963,0.00019417547,0.0007166824,0.0011608881,0.00038598897],"domain_scores_gemma":[0.96764404,0.025900107,0.0010276069,0.0015103014,0.0033505985,0.00056737615],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006712706,0.0020812706,0.0019085248,0.001727962,0.0013872214,0.0022958587,0.0022321227,0.002526441,0.022015912],"category_scores_gemma":[0.055040926,0.0009381562,0.0029121325,0.0016213986,0.005178094,0.007094299,0.004969806,0.008624305,0.0041456474],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006412835,0.000103458086,0.00080199743,0.0003788393,0.000090054295,0.00020903732,0.00031598873,0.026859587,0.000891291,0.9273881,0.013993915,0.028903631],"study_design_scores_gemma":[0.00006118927,0.000055662535,0.00028405807,0.00012864122,0.000022546756,0.00017675167,0.00003809142,0.07122573,0.0005255175,0.91929436,0.008155571,0.000031882773],"about_ca_topic_score_codex":0.0032510245,"about_ca_topic_score_gemma":0.0029162837,"teacher_disagreement_score":0.022015912,"about_ca_system_score_codex":0.002182004,"about_ca_system_score_gemma":0.0030689002,"threshold_uncertainty_score":0.07365054},"labels":[],"label_agreement":null},{"id":"W3191161744","doi":"10.1109/iwcmc51323.2021.9498960","title":"Safe Driving of Autonomous Vehicles through State Representation Learning","year":2021,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Autoencoder; Computer science; Reinforcement learning; Representation (politics); Artificial intelligence; Perception; Scheme (mathematics); Object (grammar); Function (biology); Deep learning; Simulation; Mathematics","score_opus":0.023431499770939335,"score_gpt":0.2786308688831665,"score_spread":0.2551993691122272,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3191161744","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023407457,0.000108051485,0.9742063,0.0001619323,0.000025702466,0.000023536499,0.000019420566,0.0006260259,0.0014215723],"genre_scores_gemma":[0.93664324,0.00007840288,0.06103907,0.00008365483,0.000021288763,0.000061059414,0.00005969841,0.000059120393,0.0019543623],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99963486,0.00008713781,0.000012963521,0.00010821747,0.00009571988,0.00006107842],"domain_scores_gemma":[0.9994411,0.00023347342,0.00010638759,0.000061708975,0.00011543309,0.0000418703],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000673694,0.00064498413,0.00061253615,0.0003476123,0.00032867075,0.0007609987,0.0010392081,0.0007849943,0.0009019785],"category_scores_gemma":[0.0017958072,0.0004923475,0.00046279986,0.0002445868,0.00094300037,0.0011047565,0.0011732834,0.0012316058,0.00021999699],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000031339783,0.00002474799,0.00038883238,0.000020893667,0.000014734076,0.000039360475,0.000049714043,0.9656991,0.002143322,0.005834261,0.00030807848,0.025445526],"study_design_scores_gemma":[0.0000019136876,0.000011047677,0.000031234405,0.0000012978502,0.0000012837663,0.000003778071,0.000002693772,0.99813324,0.00024760567,0.0014493633,0.00011479536,0.000001747318],"about_ca_topic_score_codex":0.0059682275,"about_ca_topic_score_gemma":0.0041572466,"teacher_disagreement_score":0.0059682275,"about_ca_system_score_codex":0.0007417617,"about_ca_system_score_gemma":0.001302439,"threshold_uncertainty_score":0.011866927},"labels":[],"label_agreement":null},{"id":"W3191305580","doi":"10.1109/isscs52333.2021.9497411","title":"Inverted Pendulum Control with a Robotic Arm using Deep Reinforcement Learning","year":2021,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Inverted pendulum; Reinforcement learning; Pendulum; Computer science; Benchmark (surveying); Double inverted pendulum; Robot; Control theory (sociology); Position (finance); Inertial measurement unit; Artificial intelligence; Simulation; Control (management); Engineering; Physics; Mechanical engineering; Nonlinear system; Geology","score_opus":0.019758795675198365,"score_gpt":0.23581916117171545,"score_spread":0.21606036549651708,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3191305580","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.097255506,0.00024114829,0.8952125,0.00025825386,0.00009500285,0.00010111862,0.00003086215,0.00089335185,0.005912364],"genre_scores_gemma":[0.9513046,0.00004870878,0.046703517,0.000064986394,0.000012537182,0.00007436146,0.000025905032,0.000020513684,0.0017448468],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998271,0.00003897102,0.000009388291,0.000034587843,0.00005297653,0.0000370772],"domain_scores_gemma":[0.999566,0.00017817589,0.000065277636,0.000037282858,0.00011269534,0.00004053092],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00064131984,0.00063182885,0.00042653465,0.00020801713,0.00029273692,0.00040743133,0.00063033536,0.000619761,0.0014667964],"category_scores_gemma":[0.0013071073,0.0002350587,0.00030062065,0.00013256301,0.0005917566,0.00035076233,0.0006140434,0.0008495569,0.00018114931],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010267885,0.0000951654,0.0006632642,0.000052492513,0.000027972868,0.00009694738,0.000049383405,0.94275206,0.0068057915,0.0035844168,0.0006041373,0.045165606],"study_design_scores_gemma":[0.000008650456,0.00005560492,0.000063287705,0.0000027178032,0.0000025352333,0.0000053450226,0.0000021946141,0.9983368,0.0006626321,0.00066896563,0.00018881932,0.0000023650002],"about_ca_topic_score_codex":0.005458185,"about_ca_topic_score_gemma":0.0042448994,"teacher_disagreement_score":0.005458185,"about_ca_system_score_codex":0.0005994173,"about_ca_system_score_gemma":0.0007931734,"threshold_uncertainty_score":0.010852814},"labels":[],"label_agreement":null},{"id":"W3193711978","doi":"10.65109/lixm5646","title":"Diverse Auto-Curriculum is Critical for Successful Real-World Multiagent Learning Systems","year":2021,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Huawei Technologies (Canada)","funders":"","keywords":"Curriculum; Cornerstone; Computer science; Diversity (politics); Benchmark (surveying); Variety (cybernetics); Process (computing); Reinforcement learning; Component (thermodynamics); Human–computer interaction; Artificial intelligence; Psychology; Pedagogy; Sociology","score_opus":0.02883001515941496,"score_gpt":0.3124728168870807,"score_spread":0.2836428017276657,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3193711978","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14061964,0.00059095933,0.83621025,0.004081617,0.00006448489,0.00023219176,0.000049418144,0.0006993883,0.017452084],"genre_scores_gemma":[0.86958903,0.00032415314,0.12704466,0.00035740732,0.000037384398,0.0002150339,0.00005077087,0.00009332797,0.0022882656],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99858,0.00056440715,0.00009400581,0.00026809354,0.00031376176,0.00017970167],"domain_scores_gemma":[0.9942624,0.0024343703,0.0007240054,0.0010667446,0.00058028026,0.00093226484],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028167684,0.00056074123,0.00070668,0.00030139892,0.0012630497,0.0016345786,0.0010797746,0.0013069765,0.0019939751],"category_scores_gemma":[0.010716665,0.00054067234,0.00043684963,0.00022841997,0.0021734734,0.00401439,0.004058782,0.0025238672,0.00052234676],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022545752,0.0005320587,0.010126363,0.00049522833,0.00015643072,0.00039330975,0.002478995,0.5515873,0.023827335,0.21279864,0.003027601,0.19435132],"study_design_scores_gemma":[0.00009050851,0.00035742603,0.002898864,0.00012107647,0.000039638177,0.0003459861,0.00074537046,0.6945859,0.0062517975,0.27765295,0.016844977,0.00006535978],"about_ca_topic_score_codex":0.00088534586,"about_ca_topic_score_gemma":0.0012654503,"teacher_disagreement_score":0.0028167684,"about_ca_system_score_codex":0.000835851,"about_ca_system_score_gemma":0.0015248278,"threshold_uncertainty_score":0.014896691},"labels":[],"label_agreement":null},{"id":"W3194918255","doi":"10.1016/j.eng.2021.04.027","title":"Actor–Critic Reinforcement Learning and Application in Developing Computer-Vision-Based Interface Tracking","year":2021,"lang":"en","type":"article","venue":"Engineering","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Computer science; Interface (matter); Context (archaeology); Process (computing); Artificial intelligence; Noise (video); Tracking (education); Control (management); Object (grammar); Machine learning; Computer vision","score_opus":0.009053442320821347,"score_gpt":0.251530277148817,"score_spread":0.24247683482799565,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3194918255","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0019352196,0.00028087787,0.99561405,0.00008926463,0.000026640893,0.000025151941,0.0000055380574,0.00025123332,0.0017720356],"genre_scores_gemma":[0.5239227,0.0011299419,0.46820906,0.00022775125,0.000080625636,0.00033104545,0.000072389135,0.00017454928,0.0058519603],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99957186,0.0001426343,0.000026719568,0.00009273223,0.000133525,0.000032547578],"domain_scores_gemma":[0.998393,0.0010658952,0.00013197487,0.00008323154,0.00026965502,0.000056154295],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015417624,0.0007859655,0.0007558942,0.00041104032,0.0002572077,0.00070448086,0.0010253629,0.0011175808,0.0019567898],"category_scores_gemma":[0.004327946,0.00047315628,0.0005318449,0.00042981998,0.0012316123,0.0006648876,0.00074272725,0.0017074895,0.0004819462],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000018099565,0.000028652656,0.00033645067,0.00008362278,0.000025215539,0.000058955942,0.000057551246,0.941246,0.0014409873,0.019155059,0.0005802944,0.036969118],"study_design_scores_gemma":[0.0000036455326,0.000010698938,0.000023863124,0.0000073025367,0.000002339975,0.000007396696,0.000002289378,0.99655837,0.00049474434,0.0022009101,0.0006859608,0.0000024867818],"about_ca_topic_score_codex":0.0043283445,"about_ca_topic_score_gemma":0.00291495,"teacher_disagreement_score":0.0043283445,"about_ca_system_score_codex":0.0009986241,"about_ca_system_score_gemma":0.0011114972,"threshold_uncertainty_score":0.008606315},"labels":[],"label_agreement":null},{"id":"W3196801835","doi":"10.48550/arxiv.2109.00157","title":"A Survey of Exploration Methods in Reinforcement Learning","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Reinforcement; Computer science; Artificial intelligence; Machine learning; Error-driven learning; Component (thermodynamics); Psychology; Social psychology","score_opus":0.18719629604214041,"score_gpt":0.27015657521499553,"score_spread":0.08296027917285512,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3196801835","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0033935779,0.107358865,0.87134564,0.0015815411,0.0004129966,0.00009126398,0.0001269598,0.00042036676,0.01526891],"genre_scores_gemma":[0.24831554,0.2041927,0.52491915,0.0014678617,0.002522926,0.00090660783,0.0005881563,0.0005723597,0.016514642],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99841356,0.0005233974,0.00013974098,0.00024399678,0.0005949505,0.000084339525],"domain_scores_gemma":[0.9982951,0.0011909717,0.000097673124,0.00012314066,0.00022380895,0.000069162925],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018643518,0.0012801809,0.0016282164,0.0010825122,0.00047850612,0.0015738006,0.0012445919,0.001341335,0.003954363],"category_scores_gemma":[0.0043636183,0.0006349074,0.0012232885,0.0021018772,0.0011669991,0.0023618347,0.0016181421,0.0022531257,0.0014607677],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011548679,0.00020985116,0.0017773467,0.0027218105,0.00015257129,0.00009731013,0.00027301346,0.1107439,0.0016856154,0.22969835,0.012012813,0.64051193],"study_design_scores_gemma":[0.00007794388,0.0003448711,0.0013157956,0.0011776568,0.00009374037,0.00035620588,0.000117000476,0.48240623,0.0019320415,0.36712858,0.14496216,0.00008781101],"about_ca_topic_score_codex":0.0020429865,"about_ca_topic_score_gemma":0.0014035512,"teacher_disagreement_score":0.003954363,"about_ca_system_score_codex":0.0012200988,"about_ca_system_score_gemma":0.0014326342,"threshold_uncertainty_score":0.013228595},"labels":[],"label_agreement":null},{"id":"W3197466741","doi":"10.48550/arxiv.2109.03331","title":"CyGIL: A Cyber Gym for Training Autonomous Agents over Emulated Network Systems","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Training (meteorology); Computer science; Cyber-physical system; Computer network; Computer security; Operating system; Physics","score_opus":0.1234854113982818,"score_gpt":0.21512399704081328,"score_spread":0.09163858564253148,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3197466741","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20110612,0.00029095,0.7431729,0.0010856529,0.00036005976,0.0010382718,0.0006331944,0.017781483,0.0345313],"genre_scores_gemma":[0.8115653,0.0001362911,0.17806116,0.00028215622,0.00002140114,0.00077022065,0.0005392791,0.000424666,0.0081994925],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996692,0.00013299988,0.000014228927,0.000059237238,0.000074776624,0.000049637463],"domain_scores_gemma":[0.9992607,0.0002778107,0.0000705861,0.00013042512,0.00007058442,0.00018969778],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007404425,0.0007303377,0.0003610138,0.0003453697,0.00048873894,0.00070173625,0.0017501011,0.0009023744,0.0068359612],"category_scores_gemma":[0.001897413,0.00023063751,0.0002449574,0.00016069852,0.001069217,0.0012472752,0.0021491917,0.001269796,0.0008959903],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00080436806,0.00086391077,0.0023527232,0.00034919285,0.000094079034,0.00046840604,0.00048010814,0.83209324,0.035501905,0.027585257,0.012141322,0.08726552],"study_design_scores_gemma":[0.0001762464,0.00071923103,0.00076952274,0.000040895575,0.000014636715,0.00010399411,0.00008552691,0.95888984,0.011144254,0.008694753,0.01932681,0.000034193396],"about_ca_topic_score_codex":0.0029605788,"about_ca_topic_score_gemma":0.0040667737,"teacher_disagreement_score":0.0068359612,"about_ca_system_score_codex":0.0007741695,"about_ca_system_score_gemma":0.0012195812,"threshold_uncertainty_score":0.022868574},"labels":[],"label_agreement":null},{"id":"W3200155431","doi":"10.48550/arxiv.2109.05110","title":"An Empirical Comparison of Off-policy Prediction Learning Algorithms in the Four Rooms Environment","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Algorithm; Artificial intelligence; Machine learning","score_opus":0.10666805785520882,"score_gpt":0.25010017098736903,"score_spread":0.1434321131321602,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3200155431","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8961235,0.005333283,0.0853029,0.0013257992,0.00044434972,0.00036701254,0.0007156046,0.0022593655,0.008128102],"genre_scores_gemma":[0.9315037,0.000852172,0.06338205,0.0003154611,0.00007616905,0.00025628993,0.0017399633,0.00015151247,0.0017226991],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99576277,0.0017849293,0.00035692187,0.00085521874,0.00081696064,0.00042325497],"domain_scores_gemma":[0.9648861,0.027592381,0.0011752442,0.0025859997,0.0027678055,0.0009924851],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009938755,0.0015516239,0.0014660556,0.0018407572,0.0007522907,0.0013118983,0.002223736,0.0026906277,0.0013652783],"category_scores_gemma":[0.035577703,0.0004267539,0.0008915319,0.0014963022,0.0012383024,0.0032565657,0.0014566671,0.0029765365,0.0005740536],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031781865,0.0021737788,0.028156286,0.0004970259,0.0002889599,0.00012697151,0.00018111532,0.66374934,0.0015430115,0.0040142797,0.007912196,0.28817886],"study_design_scores_gemma":[0.00015084825,0.0009026962,0.0044885846,0.000049343285,0.000047202113,0.00006584072,0.00012907143,0.9885356,0.0021222185,0.0024038074,0.0010761179,0.00002877872],"about_ca_topic_score_codex":0.008384654,"about_ca_topic_score_gemma":0.0053389873,"teacher_disagreement_score":0.009938755,"about_ca_system_score_codex":0.001909296,"about_ca_system_score_gemma":0.0018446636,"threshold_uncertainty_score":0.05256182},"labels":[],"label_agreement":null},{"id":"W3201099806","doi":"10.1109/lra.2021.3139145","title":"Learning Selective Communication for Multi-Agent Path Finding","year":2021,"lang":"en","type":"article","venue":"IEEE Robotics and Automation Letters","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":75,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Simon Fraser University","funders":"","keywords":"Computer science; Overhead (engineering); Imitation; Reinforcement learning; Artificial intelligence; Path (computing); Focus (optics); Simple (philosophy); Scheme (mathematics); Machine learning; Telecommunications network; Distributed computing; Computer network; Mathematics","score_opus":0.03388270364324859,"score_gpt":0.28015295791465555,"score_spread":0.24627025427140697,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3201099806","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027211785,0.00016004099,0.96891755,0.0003447011,0.000042343923,0.000073679155,0.000040397033,0.00071629224,0.0024932185],"genre_scores_gemma":[0.8458142,0.0001472857,0.14935338,0.00021608426,0.000050578838,0.00033764422,0.00013572529,0.00011039408,0.0038345803],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991672,0.00022894751,0.000049023027,0.00024213151,0.00016714972,0.00014549833],"domain_scores_gemma":[0.9964001,0.0023593223,0.00041883424,0.00031341353,0.00029725736,0.0002109608],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016307032,0.0009836124,0.0012684042,0.00058445544,0.00080066174,0.0008592272,0.002040206,0.0015302226,0.0030288456],"category_scores_gemma":[0.0071399286,0.00063381606,0.0006350814,0.0005231675,0.0013398392,0.0016041687,0.0021774678,0.0019769263,0.0004196239],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000869358,0.000062823565,0.0006811466,0.0000693389,0.000036146866,0.00008520042,0.000111292095,0.938713,0.0013027455,0.016014649,0.0010893726,0.04174741],"study_design_scores_gemma":[0.00001048232,0.000015835049,0.000028827555,0.0000023928517,0.00000343297,0.0000062322674,0.000006576826,0.99443936,0.0002296452,0.0050628977,0.00019169284,0.0000027366525],"about_ca_topic_score_codex":0.0042462903,"about_ca_topic_score_gemma":0.004051891,"teacher_disagreement_score":0.0042462903,"about_ca_system_score_codex":0.0011166757,"about_ca_system_score_gemma":0.0020017656,"threshold_uncertainty_score":0.010132492},"labels":[],"label_agreement":null},{"id":"W3201820699","doi":"10.1109/tpami.2022.3215769","title":"Continuous-Time Fitted Value Iteration for Robust Policies","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Pattern Analysis and Machine Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Toronto","funders":"","keywords":"Bellman equation; Discretization; Reinforcement learning; Leverage (statistics); Optimal control; Mathematical optimization; Computer science; Robustness (evolution); Dynamic programming; Mathematics; Artificial intelligence","score_opus":0.020695523465835194,"score_gpt":0.26019000160666267,"score_spread":0.23949447814082747,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3201820699","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007931715,0.00026619894,0.9891431,0.00017339978,0.00003683805,0.000035928733,0.000022773273,0.00025919906,0.00213091],"genre_scores_gemma":[0.7042421,0.0003548255,0.28856042,0.0002477835,0.000056074874,0.00032969768,0.000184301,0.00032077695,0.0057040746],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989122,0.00043219453,0.00006144542,0.00021183542,0.00024250898,0.00013979198],"domain_scores_gemma":[0.9957391,0.0032109986,0.00029099907,0.0002098532,0.00040165495,0.00014725163],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002439415,0.0013015592,0.0013936675,0.00062884577,0.00041764526,0.001374519,0.001062575,0.0015757912,0.0037572393],"category_scores_gemma":[0.010887995,0.0007276162,0.0009042826,0.00050998974,0.0019800598,0.0012730386,0.001670445,0.0028367862,0.00070833054],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007382223,0.00003085338,0.00036944592,0.000062617946,0.00002935331,0.000049347967,0.0000692503,0.9463266,0.00073423114,0.032260645,0.00062669924,0.01936714],"study_design_scores_gemma":[0.0000060425623,0.000012740921,0.00001592439,0.0000069353478,0.0000017732214,0.0000058440655,0.0000037180314,0.9920101,0.000219211,0.0074869236,0.00022751467,0.0000032148478],"about_ca_topic_score_codex":0.0036477698,"about_ca_topic_score_gemma":0.0024278066,"teacher_disagreement_score":0.0037572393,"about_ca_system_score_codex":0.0016689133,"about_ca_system_score_gemma":0.0021192026,"threshold_uncertainty_score":0.012901008},"labels":[],"label_agreement":null},{"id":"W3202097587","doi":"","title":"Learning One Representation to Optimize All Rewards","year":2021,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Representation (politics); Markov decision process; Artificial intelligence; Reinforcement learning; A priori and a posteriori; Machine learning; Markov process; Temporal difference learning; Mathematics","score_opus":0.04197470961659992,"score_gpt":0.29183757204241134,"score_spread":0.2498628624258114,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3202097587","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012702084,0.00015448185,0.983266,0.00040438538,0.00004695354,0.000035016717,0.00014897903,0.00066056073,0.0025815745],"genre_scores_gemma":[0.68298054,0.00038175503,0.30740863,0.0002911112,0.00010332789,0.0003181518,0.00049963844,0.0003033586,0.0077135325],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99948657,0.00014567398,0.00002371974,0.00016205234,0.00009086708,0.0000910508],"domain_scores_gemma":[0.9992404,0.0003300628,0.00008532734,0.00016994601,0.00010500889,0.00006928456],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00079356006,0.00115237,0.0010860204,0.00048408247,0.00033575416,0.0012098748,0.0014930478,0.001535766,0.00418278],"category_scores_gemma":[0.003573019,0.00045297667,0.0006503341,0.0004938149,0.0009367267,0.0022368832,0.0015212672,0.0022070012,0.0009996616],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011614713,0.00011567554,0.0006039432,0.000088528424,0.00004630242,0.000065873995,0.00006557727,0.85052574,0.0020530093,0.05762158,0.0029846893,0.08571283],"study_design_scores_gemma":[0.000013004373,0.000036445796,0.000049434995,0.000009157401,0.000007690132,0.000015576581,0.0000049503096,0.97423404,0.00070407626,0.024303775,0.00061440555,0.0000074330455],"about_ca_topic_score_codex":0.0021553023,"about_ca_topic_score_gemma":0.002631182,"teacher_disagreement_score":0.00418278,"about_ca_system_score_codex":0.0010288483,"about_ca_system_score_gemma":0.0017487652,"threshold_uncertainty_score":0.013992786},"labels":[],"label_agreement":null},{"id":"W3202849004","doi":"10.1109/iccv48922.2021.01565","title":"Interpretation of Emergent Communication in Heterogeneous Collaborative Embodied Agents","year":2021,"lang":"en","type":"article","venue":"2021 IEEE/CVF International Conference on Computer Vision (ICCV)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Western Canada Research Grid; Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research; National Science Foundation","keywords":"Embodied cognition; Computer science; Human–computer interaction; Task (project management); Perception; Oracle; Object (grammar); Models of communication; Embodied agent; Artificial intelligence; Communication; Psychology; Engineering; Software engineering","score_opus":0.03512853083241999,"score_gpt":0.3267310249632774,"score_spread":0.29160249413085737,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3202849004","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24141882,0.00044557836,0.7184965,0.0013808437,0.00008719405,0.00007099225,0.000109584464,0.00026878083,0.03772179],"genre_scores_gemma":[0.9606385,0.00016941395,0.03705222,0.00006535433,0.00002891953,0.00006524511,0.000051559848,0.000036758345,0.0018919699],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992787,0.00031638413,0.000040056333,0.00013772695,0.00014437418,0.00008280376],"domain_scores_gemma":[0.99800223,0.0009979788,0.0003286719,0.00027310554,0.00026703565,0.00013082719],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00082977617,0.00033505645,0.0003227991,0.00076017313,0.00092381856,0.0018591683,0.0008947618,0.0010084219,0.0024136638],"category_scores_gemma":[0.00449572,0.0002475919,0.00043535038,0.00044783935,0.00306382,0.0028857135,0.0022079214,0.000787227,0.0001643004],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000078502184,0.000031161573,0.0020576322,0.00011316233,0.000027454447,0.0008855667,0.005994971,0.05408781,0.005647825,0.9139423,0.00074998586,0.016383715],"study_design_scores_gemma":[0.000034446355,0.00003871106,0.0016203553,0.000040922958,0.000019533352,0.00021539922,0.0024098377,0.31908864,0.0014956495,0.6699117,0.005096093,0.000028746526],"about_ca_topic_score_codex":0.001717571,"about_ca_topic_score_gemma":0.000854778,"teacher_disagreement_score":0.0024136638,"about_ca_system_score_codex":0.0011583897,"about_ca_system_score_gemma":0.0005010969,"threshold_uncertainty_score":0.008404732},"labels":[],"label_agreement":null},{"id":"W3202938653","doi":"10.1109/lra.2022.3224667","title":"Deep Reinforcement Learning for Decentralized Multi-Robot Exploration With Macro Actions","year":2022,"lang":"en","type":"article","venue":"IEEE Robotics and Automation Letters","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Reinforcement learning; Computer science; Robot; Robustness (evolution); Scalability; Artificial intelligence; Benchmark (surveying); Action selection; Macro; Computation; Machine learning; Distributed computing","score_opus":0.03577697130002772,"score_gpt":0.26641158264475134,"score_spread":0.23063461134472363,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3202938653","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.068661615,0.00026601058,0.9283103,0.00022083758,0.000047570964,0.00004276851,0.000041338804,0.00069661054,0.0017129191],"genre_scores_gemma":[0.94774646,0.00006324765,0.050477978,0.00008187461,0.0000148472745,0.00006968065,0.00006717477,0.00003605317,0.0014427877],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997658,0.000071248,0.000010525098,0.00005376384,0.000053056352,0.000045683817],"domain_scores_gemma":[0.99923265,0.0003834159,0.00010419447,0.000088330926,0.0001169928,0.00007437515],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008347955,0.0005455974,0.00066942,0.00019141761,0.00028479856,0.00043102237,0.0010501066,0.0005345927,0.0009473379],"category_scores_gemma":[0.0022184036,0.00031684383,0.00025846437,0.0001630615,0.00064651994,0.00072192284,0.0009985418,0.0012097302,0.00016855115],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000066994784,0.000059972015,0.0008554871,0.000030999425,0.000025117655,0.000046459692,0.0000415645,0.95695746,0.0019837532,0.003060345,0.0005906766,0.036281127],"study_design_scores_gemma":[0.0000039210727,0.000014769536,0.00004987087,0.0000012739977,0.00000126131,0.0000030846463,0.0000026393138,0.99854016,0.00022280193,0.0010726064,0.00008632385,0.0000011930824],"about_ca_topic_score_codex":0.004721893,"about_ca_topic_score_gemma":0.00537821,"teacher_disagreement_score":0.004721893,"about_ca_system_score_codex":0.0008875368,"about_ca_system_score_gemma":0.0010816369,"threshold_uncertainty_score":0.009388804},"labels":[],"label_agreement":null},{"id":"W3203207428","doi":"10.1007/s00521-021-06259-1","title":"Policy invariant explicit shaping: an efficient alternative to reward shaping","year":2021,"lang":"en","type":"article","venue":"Neural Computing and Applications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada; University of Alberta; Alberta Machine Intelligence Institute; Compute Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Advice (programming); Computer science; Invariant (physics); Function (biology); Process (computing); Temporal difference learning; Artificial intelligence; Mathematics","score_opus":0.06130619872746178,"score_gpt":0.3291911299996931,"score_spread":0.2678849312722313,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3203207428","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016726675,0.00018922909,0.97671366,0.00028852563,0.000085518666,0.00006913687,0.000030002013,0.0016642223,0.004233039],"genre_scores_gemma":[0.79028296,0.00016903429,0.20409712,0.00031108494,0.00006846709,0.00013783578,0.00006375252,0.0002757582,0.004593985],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99874616,0.00037963854,0.00006755534,0.00024563677,0.00039493482,0.00016611203],"domain_scores_gemma":[0.9964335,0.0018772659,0.0003262297,0.0007624478,0.0003730449,0.00022752794],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014564276,0.0009129381,0.0010079243,0.00046544702,0.00044463627,0.00094053015,0.0017663632,0.0014062298,0.003975269],"category_scores_gemma":[0.008594183,0.00037876805,0.0004498267,0.0004129979,0.0013327923,0.0012213844,0.0017613382,0.0022955637,0.00087821396],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051589607,0.00023551706,0.00095441286,0.00015861301,0.00005240942,0.00023045295,0.00020175066,0.64257854,0.018897267,0.09218501,0.00385694,0.24013315],"study_design_scores_gemma":[0.0000226968,0.000054438948,0.000055195927,0.000010468179,0.000006895987,0.000025901532,0.000006497336,0.9821879,0.0021238667,0.014402805,0.0010938908,0.000009298115],"about_ca_topic_score_codex":0.0017080887,"about_ca_topic_score_gemma":0.0016048476,"teacher_disagreement_score":0.003975269,"about_ca_system_score_codex":0.00074626715,"about_ca_system_score_gemma":0.001587469,"threshold_uncertainty_score":0.013298631},"labels":[],"label_agreement":null},{"id":"W3204293975","doi":"10.1007/s13235-023-00490-2","title":"Robustness and Sample Complexity of Model-Based MARL for General-Sum Markov Games","year":2023,"lang":"en","type":"article","venue":"Dynamic Games and Applications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Ministère de la Défense Nationale","keywords":"Markov chain; Mathematics; Markov kernel; Markov perfect equilibrium; Markov process; Markov model; Discrete mathematics; Applied mathematics; Mathematical optimization; Combinatorics; Mathematical economics; Variable-order Markov model; Nash equilibrium; Statistics","score_opus":0.03328730004061116,"score_gpt":0.28858203258423304,"score_spread":0.2552947325436219,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3204293975","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14107211,0.0009144432,0.84632456,0.0019311144,0.0000937364,0.00019976533,0.00060932274,0.000711089,0.008143879],"genre_scores_gemma":[0.9538854,0.00043371855,0.039965264,0.00036872565,0.00016191353,0.00032343756,0.0006690613,0.00033004567,0.0038624306],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9935184,0.0031005223,0.00028073715,0.0012688981,0.0010953707,0.0007360735],"domain_scores_gemma":[0.85479075,0.12762724,0.006571738,0.005080716,0.0035481618,0.002381394],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012338568,0.0023214866,0.00389913,0.00246063,0.0015441623,0.004371957,0.0043785893,0.0031952353,0.004917636],"category_scores_gemma":[0.07805885,0.0017985274,0.001807843,0.0011666424,0.005416887,0.0072019314,0.0056911204,0.005689342,0.0005708925],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042352785,0.0001430576,0.0013419428,0.00017450043,0.00010545307,0.00008807086,0.000118565426,0.8952778,0.0007886318,0.09372538,0.0011850749,0.006628063],"study_design_scores_gemma":[0.000013653568,0.000028155868,0.00012162864,0.000010905867,0.000007898478,0.000010596833,0.000009593284,0.95982134,0.00016338853,0.039743327,0.00005679744,0.000012639813],"about_ca_topic_score_codex":0.005262939,"about_ca_topic_score_gemma":0.0042679487,"teacher_disagreement_score":0.012338568,"about_ca_system_score_codex":0.0048951097,"about_ca_system_score_gemma":0.0039534746,"threshold_uncertainty_score":0.06525332},"labels":[],"label_agreement":null},{"id":"W3204811657","doi":"","title":"LOCO: Adaptive exploration in reinforcement learning via local estimation of contraction coefficients","year":2021,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Reinforcement learning; Contraction (grammar); Computer science; Reinforcement; Artificial intelligence; Control theory (sociology); Mathematics; Mathematical optimization; Engineering; Structural engineering","score_opus":0.053273948608731185,"score_gpt":0.33253110950232107,"score_spread":0.2792571608935899,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3204811657","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012319522,0.00017457332,0.9832496,0.00012634062,0.00007483478,0.000055431185,0.000057298304,0.0024439362,0.001498313],"genre_scores_gemma":[0.67833817,0.00015581098,0.31404257,0.00024248958,0.0000816809,0.00048724964,0.00022307615,0.0008891205,0.005539875],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996768,0.000100600395,0.000014257507,0.00007497714,0.000090816786,0.00004247815],"domain_scores_gemma":[0.9990753,0.0004978036,0.0000757314,0.00013389683,0.00011961766,0.00009769336],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012267078,0.0007911137,0.0011569175,0.00046702076,0.00035917156,0.00086746627,0.0017544375,0.0013636461,0.0051996335],"category_scores_gemma":[0.0039471146,0.0004943167,0.00039948017,0.00039456828,0.0009939972,0.0015003041,0.0024805428,0.0016245464,0.00094831357],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006252609,0.00028122606,0.0013622965,0.0002771499,0.00009247206,0.00019524501,0.00018460024,0.5905268,0.011293349,0.03340147,0.008827946,0.35293218],"study_design_scores_gemma":[0.00002491328,0.000038206537,0.000051117127,0.000005715662,0.0000035316118,0.00001362734,0.000004604035,0.99568033,0.0008137893,0.0029353693,0.0004236074,0.0000051770585],"about_ca_topic_score_codex":0.0021249873,"about_ca_topic_score_gemma":0.0025998645,"teacher_disagreement_score":0.0051996335,"about_ca_system_score_codex":0.00048746876,"about_ca_system_score_gemma":0.0008720825,"threshold_uncertainty_score":0.017394543},"labels":[],"label_agreement":null},{"id":"W3205279237","doi":"10.1109/icra48506.2021.9560793","title":"Continual Model-Based Reinforcement Learning with Hypernetworks","year":2021,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Toronto","funders":"Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; National Science Foundation","keywords":"Reinforcement learning; Computer science; Task (project management); Artificial intelligence; Machine learning; Dynamics (music); Control (management); Robot; Plan (archaeology); State (computer science); Engineering","score_opus":0.01245078320415947,"score_gpt":0.22087272369292985,"score_spread":0.20842194048877039,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3205279237","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06539355,0.00042391202,0.9277636,0.00047414718,0.00008017661,0.00008318424,0.00017546,0.0015615837,0.0040444587],"genre_scores_gemma":[0.92372924,0.00016394716,0.07202865,0.00017683435,0.00004170854,0.00022305573,0.0002152664,0.000113371985,0.0033079141],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994454,0.0001974575,0.000031092684,0.00014492981,0.00011174557,0.000069386464],"domain_scores_gemma":[0.9969573,0.0020352746,0.00022018785,0.0002846406,0.00033233105,0.00017026972],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00143667,0.0009498391,0.0009830638,0.00057353463,0.0004082281,0.000930621,0.0022119177,0.001063132,0.0037224055],"category_scores_gemma":[0.0054534655,0.0006869199,0.0005617317,0.0004279583,0.0013430006,0.0023087359,0.0014944111,0.0023075116,0.0004764848],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009314675,0.000057904683,0.00040658418,0.000034856228,0.000026710164,0.000040749426,0.000045993067,0.96749127,0.0005314327,0.0055058156,0.00061997806,0.025145574],"study_design_scores_gemma":[0.000008409462,0.000013140299,0.000024290737,0.0000026383882,0.000002376039,0.0000038806547,0.000002455796,0.9963716,0.00014005174,0.00330974,0.00011877086,0.00000262377],"about_ca_topic_score_codex":0.0071975756,"about_ca_topic_score_gemma":0.007163219,"teacher_disagreement_score":0.0071975756,"about_ca_system_score_codex":0.001442971,"about_ca_system_score_gemma":0.0009990794,"threshold_uncertainty_score":0.014311373},"labels":[],"label_agreement":null},{"id":"W3205427743","doi":"","title":"MASAI: Multi-agent Summative Assessment Improvement for Unsupervised Environment Design","year":2021,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Summative assessment; Computer science; Formative assessment; Psychology; Mathematics education","score_opus":0.09432305776176142,"score_gpt":0.3331575425502579,"score_spread":0.2388344847884965,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3205427743","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007633956,0.000092996575,0.9864162,0.00010059577,0.00006164962,0.0001408372,0.000050311628,0.003474684,0.0020287316],"genre_scores_gemma":[0.34012812,0.000066572444,0.65317154,0.000201275,0.00005074558,0.0005002867,0.00028403715,0.000461523,0.0051358147],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980544,0.0007337182,0.00009418186,0.0003152897,0.0006455557,0.00015686195],"domain_scores_gemma":[0.99695754,0.0013714159,0.00022587173,0.00036395405,0.00087662874,0.00020453661],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003495364,0.0015247014,0.0012873342,0.0008654398,0.0008815264,0.0009030444,0.002641384,0.0013052872,0.006046774],"category_scores_gemma":[0.008879366,0.00051901315,0.00076532003,0.00046901742,0.00081158517,0.0014035212,0.0038685934,0.00208374,0.0011273088],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005055499,0.0005114389,0.0017573774,0.00025273286,0.00017066467,0.0001611816,0.00039031234,0.48438397,0.007757111,0.01174991,0.009038102,0.48332173],"study_design_scores_gemma":[0.000030228774,0.00008884027,0.00015418518,0.000011408888,0.000014186626,0.000015749505,0.000018691164,0.9928948,0.0016988312,0.0038421224,0.0012219814,0.000008917425],"about_ca_topic_score_codex":0.0040292256,"about_ca_topic_score_gemma":0.0068062777,"teacher_disagreement_score":0.006046774,"about_ca_system_score_codex":0.00082067976,"about_ca_system_score_gemma":0.0017111159,"threshold_uncertainty_score":0.020228446},"labels":[],"label_agreement":null},{"id":"W3206540493","doi":"10.1109/icra48506.2021.9561333","title":"Shaping Rewards for Reinforcement Learning with Imperfect Demonstrations using Generative Models","year":2021,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Vector Institute; University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Generative grammar; Artificial intelligence; Machine learning; Generative model; Adversarial system; Function (biology); Action (physics); State space; Convergence (economics); Bellman equation; Range (aeronautics); Imperfect; Imitation; Mathematical optimization; Engineering; Mathematics","score_opus":0.07543834377601084,"score_gpt":0.2929367701417165,"score_spread":0.21749842636570565,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3206540493","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014167962,0.00012876789,0.983597,0.00020267715,0.000019988394,0.00003212245,0.000028348395,0.0003346713,0.0014884254],"genre_scores_gemma":[0.88425684,0.00018628211,0.111010425,0.00020078992,0.000036128145,0.00021462748,0.00010058361,0.00016428712,0.0038299146],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993414,0.00026396083,0.00002558461,0.00013076358,0.00015858545,0.000079694946],"domain_scores_gemma":[0.99621624,0.002934663,0.0002611128,0.0002562779,0.0001880587,0.00014356077],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017233924,0.00093105074,0.0010810158,0.0005586173,0.0003900459,0.00081731885,0.0014735193,0.0011658237,0.0026674515],"category_scores_gemma":[0.007652349,0.00068023824,0.00068564195,0.00035869266,0.0020410577,0.0015348404,0.0019360706,0.0022937811,0.00038462286],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003417653,0.000022831638,0.00034000684,0.000027772974,0.000013999198,0.000047339843,0.000044862034,0.96722066,0.0007591182,0.017848596,0.00032843728,0.01331215],"study_design_scores_gemma":[0.0000045807096,0.000011011294,0.000027659062,0.0000041745084,0.0000022617626,0.00000807756,0.000002228877,0.9923152,0.00023690822,0.007242395,0.00014238188,0.0000031494617],"about_ca_topic_score_codex":0.0030064217,"about_ca_topic_score_gemma":0.0035404745,"teacher_disagreement_score":0.0030064217,"about_ca_system_score_codex":0.0013165264,"about_ca_system_score_gemma":0.0011780972,"threshold_uncertainty_score":0.009552181},"labels":[],"label_agreement":null},{"id":"W3206746211","doi":"10.1109/icra48506.2021.9561017","title":"Distilling a Hierarchical Policy for Planning and Control via Representation and Reinforcement Learning","year":2021,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Reinforcement learning; Computer science; Hierarchy; Task (project management); Set (abstract data type); Artificial intelligence; Representation (politics); Control (management); Latent variable; Imitation; Machine learning; Sequence (biology); State space; Human–computer interaction; Engineering; Mathematics","score_opus":0.02105963226643383,"score_gpt":0.2996378555369776,"score_spread":0.27857822327054377,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3206746211","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0075512747,0.00006563818,0.9908274,0.00010725009,0.0000121885605,0.000036529036,0.000028688079,0.00050256634,0.0008684375],"genre_scores_gemma":[0.64215153,0.00016041411,0.3550629,0.00012152191,0.000027672239,0.000321444,0.00016763914,0.00010897579,0.0018779726],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99935883,0.00019583349,0.000034356068,0.0001658091,0.00015285383,0.00009238599],"domain_scores_gemma":[0.99908304,0.0004895854,0.00011728761,0.00012799955,0.000116404735,0.0000656548],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011415564,0.0008899278,0.00079787226,0.0004597194,0.00038351695,0.00079434155,0.0012684036,0.0009034648,0.0020508626],"category_scores_gemma":[0.0031194256,0.00051691785,0.00064023584,0.00043835217,0.001267702,0.0014051414,0.0012849363,0.0017170028,0.00045441813],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006255886,0.000055750526,0.0003832869,0.00005213268,0.00002054365,0.00005290045,0.00009626417,0.933231,0.0026208123,0.025873967,0.0007658131,0.036785003],"study_design_scores_gemma":[0.000008120094,0.000016189475,0.0000308751,0.0000035981805,0.0000027171839,0.000004616729,0.0000036947743,0.99259055,0.00038770973,0.006740368,0.00020761912,0.000004021156],"about_ca_topic_score_codex":0.008594135,"about_ca_topic_score_gemma":0.007819917,"teacher_disagreement_score":0.008594135,"about_ca_system_score_codex":0.0014045386,"about_ca_system_score_gemma":0.0025873173,"threshold_uncertainty_score":0.017088175},"labels":[],"label_agreement":null},{"id":"W3206893708","doi":"","title":"An Independent Learning Algorithm for a Class of Symmetric Stochastic Games","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Class (philosophy); Property (philosophy); Reinforcement learning; Mathematical economics; Computer science; Action (physics); Symmetry (geometry); Zero (linguistics); Mathematics; Mathematical optimization; Artificial intelligence; Discrete mathematics","score_opus":0.0524186279386688,"score_gpt":0.2089538804048337,"score_spread":0.1565352524661649,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3206893708","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015408619,0.000053365584,0.98235637,0.00013743092,0.000020486901,0.00013109209,0.000026705793,0.00017159298,0.0016944238],"genre_scores_gemma":[0.47031218,0.00017374771,0.52238685,0.00027690685,0.00006204656,0.0008629182,0.00026317165,0.00016026842,0.0055018584],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99854726,0.0004987306,0.00007488478,0.00041505604,0.00032017927,0.00014379981],"domain_scores_gemma":[0.9948277,0.0035285612,0.00040096097,0.0004289799,0.0005332772,0.0002804955],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031980446,0.0011921704,0.0014072935,0.00075421983,0.00066845067,0.0010378867,0.0031479297,0.0017314983,0.0028070654],"category_scores_gemma":[0.01158019,0.00066660985,0.001001323,0.0005659568,0.0017866122,0.0024725902,0.0023489236,0.003147786,0.00073803565],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026554038,0.00028575372,0.0014846602,0.00012638977,0.00012552826,0.00014925428,0.00027741838,0.6849149,0.0024698284,0.17817445,0.0027228016,0.12900347],"study_design_scores_gemma":[0.00004270717,0.000054807868,0.000049737504,0.000007787908,0.000007883407,0.000030379986,0.000008162467,0.96455383,0.00051831204,0.034345794,0.0003728771,0.00000768607],"about_ca_topic_score_codex":0.0014319897,"about_ca_topic_score_gemma":0.0015317505,"teacher_disagreement_score":0.0031980446,"about_ca_system_score_codex":0.0012512095,"about_ca_system_score_gemma":0.0024379012,"threshold_uncertainty_score":0.016913056},"labels":[],"label_agreement":null},{"id":"W3208016380","doi":"10.3389/frai.2022.805823","title":"Investigation of independent reinforcement learning algorithms in multi-agent environments","year":2022,"lang":"en","type":"article","venue":"Frontiers in Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Computer science; Reinforcement learning; Algorithm; Artificial intelligence; Machine learning","score_opus":0.05814036785717878,"score_gpt":0.2710809898253303,"score_spread":0.21294062196815153,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3208016380","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1392541,0.00096852484,0.8523829,0.00043548425,0.00007940399,0.00022873115,0.000054025746,0.00063256925,0.0059642005],"genre_scores_gemma":[0.8585444,0.00028208684,0.13893977,0.00012534946,0.000044101107,0.000207064,0.00012278462,0.00009415767,0.0016403045],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99694425,0.001277122,0.00014111471,0.00067895453,0.0006381954,0.0003203819],"domain_scores_gemma":[0.9776764,0.015701938,0.0018949603,0.002265223,0.001482589,0.0009788751],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0067811003,0.0015255655,0.0017168109,0.0010655574,0.0008715758,0.0013790322,0.0027150242,0.0015355598,0.0016191701],"category_scores_gemma":[0.024949748,0.00083225314,0.0009963433,0.0007210267,0.0019118148,0.003121131,0.001810941,0.0026128974,0.0003367757],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024335402,0.00025649826,0.0032921324,0.00012863553,0.00013983563,0.00006373267,0.00011225559,0.93008965,0.00082385057,0.016554285,0.00057454803,0.047721166],"study_design_scores_gemma":[0.000024235722,0.00007746738,0.00022643052,0.000007700205,0.000010925209,0.00001586245,0.000010854245,0.9940818,0.0003341845,0.004995292,0.00020922297,0.0000059003723],"about_ca_topic_score_codex":0.0024627892,"about_ca_topic_score_gemma":0.0017633447,"teacher_disagreement_score":0.0067811003,"about_ca_system_score_codex":0.0012235899,"about_ca_system_score_gemma":0.0022048412,"threshold_uncertainty_score":0.035862327},"labels":[],"label_agreement":null},{"id":"W3208994216","doi":"10.48550/arxiv.2110.14096","title":"Towards Robust Bisimulation Metric Learning","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Embedding; Computer science; Robustness (evolution); Artificial intelligence; Representation (politics); Theoretical computer science","score_opus":0.09633591901691252,"score_gpt":0.19975337179240849,"score_spread":0.10341745277549597,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3208994216","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010353166,0.00017950374,0.9881247,0.00022646152,0.000019685462,0.000028606992,0.000039446073,0.00028975782,0.00073867885],"genre_scores_gemma":[0.6103864,0.00044753592,0.38434342,0.00034048443,0.000099894816,0.00037584422,0.00041737276,0.00052233506,0.0030666622],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9970155,0.001512532,0.00015445639,0.0005335918,0.00060851744,0.0001753177],"domain_scores_gemma":[0.9915804,0.005163124,0.00096405455,0.00095668045,0.000914158,0.0004214025],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046363017,0.0019155233,0.001852279,0.0012152848,0.00045663092,0.0018685226,0.0019636678,0.0018300645,0.0018176273],"category_scores_gemma":[0.022569468,0.0009954742,0.0008103641,0.00078621606,0.0022110308,0.0037542006,0.005212707,0.004476962,0.0005927837],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015524466,0.00008943938,0.0007729051,0.00014772086,0.00007874067,0.000039167786,0.00014459123,0.806215,0.0023978918,0.12214241,0.0014874834,0.066329494],"study_design_scores_gemma":[0.0000058830947,0.000029629677,0.00003708181,0.000009475499,0.000002373909,0.000005868772,0.000005496884,0.9680499,0.00041806567,0.03114974,0.00028105825,0.0000053580598],"about_ca_topic_score_codex":0.002133969,"about_ca_topic_score_gemma":0.0017088604,"teacher_disagreement_score":0.0046363017,"about_ca_system_score_codex":0.0022090783,"about_ca_system_score_gemma":0.0018950233,"threshold_uncertainty_score":0.024519384},"labels":[],"label_agreement":null},{"id":"W3209211538","doi":"10.48550/arxiv.2110.15572","title":"Understanding the Effect of Stochasticity in Policy Optimization","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Mathematical optimization; Oracle; Convergence (economics); Initialization; Stochastic optimization; Rate of convergence; Computer science; Optimization problem; Mathematics; Key (lock); Economics","score_opus":0.10616217370546102,"score_gpt":0.20939231183322496,"score_spread":0.10323013812776394,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3209211538","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044556856,0.0003393774,0.9514996,0.0007837349,0.00004705379,0.000039906085,0.000033064254,0.00014634855,0.0025539591],"genre_scores_gemma":[0.9160882,0.0005920406,0.08081082,0.00037790637,0.0000978362,0.00009241257,0.000074792384,0.0001454944,0.0017204038],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99712104,0.0013657345,0.00013665183,0.00048078276,0.00058241613,0.00031327675],"domain_scores_gemma":[0.96716446,0.026928198,0.0024436258,0.0016397453,0.0012367122,0.00058715796],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056006676,0.0009795551,0.0014095347,0.00059892103,0.00071209273,0.0017918079,0.0013308522,0.0016974622,0.0020379082],"category_scores_gemma":[0.044628054,0.0008724524,0.000809209,0.00044369296,0.0026563962,0.0037783612,0.0027816563,0.0032007652,0.0002599935],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000055187957,0.000029886083,0.00087248604,0.000055391378,0.000036572837,0.000050951807,0.00006509164,0.95295346,0.0011416038,0.03842221,0.00018646533,0.0061306185],"study_design_scores_gemma":[0.000005067217,0.000037130103,0.00016852118,0.000009274041,0.0000054087877,0.000012497748,0.000007767802,0.98332053,0.00055005465,0.015739165,0.00013842752,0.0000060517636],"about_ca_topic_score_codex":0.0051664445,"about_ca_topic_score_gemma":0.0026901185,"teacher_disagreement_score":0.0056006676,"about_ca_system_score_codex":0.001836776,"about_ca_system_score_gemma":0.002619234,"threshold_uncertainty_score":0.029619455},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"simulation_or_modeling","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"simulation_or_modeling","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"low"}],"label_agreement":"agree"},{"id":"W3209247505","doi":"10.1109/ccece53047.2021.9569056","title":"Reinforcement Learning Algorithms: An Overview and Classification","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":102,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Machine learning; Perspective (graphical); Field (mathematics); Robotics; Learning classifier system; Robot learning; Drone; Algorithm; Robot; Mobile robot","score_opus":0.0971463287431973,"score_gpt":0.32732024824265415,"score_spread":0.23017391949945687,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3209247505","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0055235447,0.3375566,0.6236008,0.0033310014,0.0009922091,0.00028449832,0.00029462835,0.0007821093,0.027634652],"genre_scores_gemma":[0.15092035,0.46178994,0.36668578,0.0015640791,0.0040190546,0.00083763525,0.0014219726,0.00039425123,0.01236698],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9982217,0.00035277745,0.00022797844,0.00037996026,0.00070935575,0.00010817007],"domain_scores_gemma":[0.99679667,0.0021051439,0.00023137475,0.00019793872,0.0005469323,0.00012191756],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024610502,0.0017274414,0.0016122436,0.0037498765,0.00058063323,0.0032097949,0.0021146808,0.002441392,0.0027789343],"category_scores_gemma":[0.005906574,0.00088844256,0.0013523599,0.0053281453,0.0015326594,0.0037310484,0.001469993,0.0036826248,0.001826853],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010808754,0.00020546984,0.0039472785,0.0022224968,0.00012727774,0.00014206037,0.00020985534,0.06290054,0.000688921,0.14223129,0.018296067,0.7689206],"study_design_scores_gemma":[0.00006116784,0.00037418708,0.0041322587,0.00215078,0.00011768965,0.0011966213,0.00022069862,0.3452923,0.0019714402,0.3777535,0.26655862,0.00017076611],"about_ca_topic_score_codex":0.0025775332,"about_ca_topic_score_gemma":0.0009697261,"teacher_disagreement_score":0.0037498765,"about_ca_system_score_codex":0.0016270533,"about_ca_system_score_gemma":0.0012969836,"threshold_uncertainty_score":0.013015389},"labels":[],"label_agreement":null},{"id":"W3209275463","doi":"10.1007/978-3-030-87136-9_4","title":"Dynamic VNF Resource Scaling and Migration: A Machine Learning Approach","year":2021,"lang":"en","type":"book-chapter","venue":"Wireless networks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Scaling; Resource (disambiguation); Distributed computing; Computer network","score_opus":0.0113441275927419,"score_gpt":0.2061103806195461,"score_spread":0.1947662530268042,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3209275463","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008982955,0.0013732554,0.98021346,0.0005728039,0.0002328829,0.000027536009,0.00004164091,0.00023394426,0.008321521],"genre_scores_gemma":[0.7454346,0.002649314,0.2228308,0.00042677892,0.0007036385,0.00017221412,0.000111596346,0.0003060408,0.027365096],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99964356,0.00012680583,0.000012397337,0.000080221784,0.000074170945,0.00006291866],"domain_scores_gemma":[0.9988953,0.0007068041,0.00011394779,0.000083149855,0.00014684025,0.00005398114],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011696556,0.00085808703,0.0011141973,0.00060163205,0.00043897337,0.0014468768,0.0021293072,0.0015077378,0.0037926896],"category_scores_gemma":[0.0037210262,0.00043209485,0.00051114755,0.0012408037,0.0010895177,0.0018286696,0.000960351,0.001746279,0.00044913683],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002152132,0.000036802052,0.00020292602,0.00004885762,0.000020487822,0.000037358688,0.000025756102,0.9055426,0.0005532364,0.034017794,0.0033376536,0.0561551],"study_design_scores_gemma":[0.0000012036907,0.0000041880685,0.000027378768,0.000004242142,0.0000024398194,0.000010154083,0.0000050390236,0.98884493,0.0000690857,0.010606602,0.00042226372,0.0000024915114],"about_ca_topic_score_codex":0.0042464253,"about_ca_topic_score_gemma":0.004226469,"teacher_disagreement_score":0.0042464253,"about_ca_system_score_codex":0.0012372131,"about_ca_system_score_gemma":0.0007202276,"threshold_uncertainty_score":0.012687862},"labels":[],"label_agreement":null},{"id":"W3210129106","doi":"10.48550/arxiv.2111.00876","title":"On the Expressivity of Markov Reward","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Set (abstract data type); Markov chain; Computer science; Task (project management); Expressivity; Reinforcement learning; Function (biology); Frame (networking); Construct (python library); Markov decision process; Markov process; Artificial intelligence; Machine learning; Cognitive psychology; Psychology; Mathematics; Programming language; Statistics","score_opus":0.06913435806116296,"score_gpt":0.18231306356787266,"score_spread":0.1131787055067097,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3210129106","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033350583,0.0003277373,0.9563185,0.0020499704,0.000041985706,0.000039602983,0.00014223631,0.00028325888,0.0074460567],"genre_scores_gemma":[0.8510544,0.00043584133,0.14473848,0.00043379428,0.00008875629,0.00016552898,0.00019021714,0.00016823335,0.0027247255],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9955142,0.0024652595,0.00025366247,0.00065060385,0.0007431727,0.0003731656],"domain_scores_gemma":[0.9799084,0.015186202,0.0014659155,0.001818751,0.0009813845,0.0006393401],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00517846,0.0009983638,0.00092114136,0.0006529067,0.00076799904,0.0026293532,0.0013417917,0.0015185862,0.0026499096],"category_scores_gemma":[0.026855607,0.00058312644,0.00093405665,0.0007611347,0.005090751,0.007879413,0.002058968,0.004448504,0.00038093296],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000114579896,0.00004874924,0.0011594043,0.00008091912,0.000031924254,0.00006195581,0.00027091603,0.16226204,0.0012293673,0.818228,0.0010382511,0.015473792],"study_design_scores_gemma":[0.00001799342,0.000051979707,0.00021316556,0.00002746085,0.000010239356,0.000034543194,0.000033496417,0.40297395,0.0005857248,0.5948443,0.0011919978,0.000015161677],"about_ca_topic_score_codex":0.0028018649,"about_ca_topic_score_gemma":0.0025082938,"teacher_disagreement_score":0.00517846,"about_ca_system_score_codex":0.0030622038,"about_ca_system_score_gemma":0.0019853436,"threshold_uncertainty_score":0.027386665},"labels":[],"label_agreement":null},{"id":"W3210485068","doi":"10.1613/jair.1.13445","title":"Multi-Agent Advisor Q-Learning","year":2022,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; University of Waterloo; Mitacs; University of Alberta; Alberta Machine Intelligence Institute; Compute Canada; Government of Canada; Vector Institute; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Computer science; Variety (cybernetics); Artificial intelligence; Action (physics); Heuristic; Sample complexity; Software deployment; Convergence (economics); Point (geometry); Operations research; Machine learning; Q-learning; Software engineering; Mathematics","score_opus":0.22346943620517123,"score_gpt":0.42782496434760386,"score_spread":0.20435552814243263,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3210485068","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018823637,0.00051168824,0.9764914,0.0005341831,0.00007576031,0.00015091716,0.000054582626,0.00074769603,0.0026101489],"genre_scores_gemma":[0.73883665,0.00029518967,0.25437722,0.0006875502,0.000101722064,0.0003318769,0.0002175454,0.00012956267,0.005022632],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99831235,0.0007972575,0.00007962548,0.00034624102,0.00027480948,0.00018980079],"domain_scores_gemma":[0.99219066,0.005717582,0.0004724927,0.00040631148,0.00079477456,0.00041811398],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004377762,0.001350266,0.0021309867,0.0006381302,0.00059779553,0.0010418752,0.0026305402,0.002106128,0.004974888],"category_scores_gemma":[0.012086235,0.000575283,0.00056389516,0.00069764006,0.0014126766,0.0015368382,0.0016925917,0.0025893336,0.00076517666],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024671902,0.00020851631,0.0017573845,0.00013278521,0.00006988092,0.00006862577,0.00009783855,0.8564205,0.0004480262,0.018699136,0.0025305515,0.11932002],"study_design_scores_gemma":[0.000029785155,0.000040485407,0.000047270663,0.00000701002,0.0000048790102,0.000009918282,0.0000056722765,0.9946773,0.00016010793,0.0046129557,0.00040090978,0.000003777016],"about_ca_topic_score_codex":0.0044673746,"about_ca_topic_score_gemma":0.0044348156,"teacher_disagreement_score":0.004974888,"about_ca_system_score_codex":0.0013072098,"about_ca_system_score_gemma":0.0022696014,"threshold_uncertainty_score":0.023152113},"labels":[],"label_agreement":null},{"id":"W3211684788","doi":"","title":"Average-Reward Learning and Planning with Options","year":2021,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Markov decision process; Computer science; Abstraction; Convergence (economics); Sample complexity; Artificial intelligence; Machine learning; Mathematical proof; Markov chain; Sample (material); Domain (mathematical analysis); Markov process; Mathematics; Statistics","score_opus":0.01646724830475379,"score_gpt":0.2526428685441239,"score_spread":0.23617562023937008,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3211684788","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009048671,0.00015585581,0.988221,0.00014584162,0.000026871834,0.000019386205,0.000032292857,0.0001688881,0.0021812622],"genre_scores_gemma":[0.6124535,0.00042271943,0.3819631,0.00016207699,0.00005837754,0.00018127593,0.00014391208,0.00011940976,0.00449563],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990107,0.00040761588,0.000055561115,0.00019756691,0.00023064543,0.0000978502],"domain_scores_gemma":[0.9972995,0.0017048833,0.00025213475,0.0003194037,0.00024055401,0.00018356422],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002110958,0.0007984567,0.0008541836,0.0005686482,0.0004563681,0.0012015468,0.0020295484,0.0009456483,0.004209871],"category_scores_gemma":[0.008165511,0.0004325076,0.00080472825,0.0008361548,0.001548346,0.0034871586,0.0019436002,0.002296028,0.00045158953],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009632625,0.000058193167,0.0005522778,0.000086194996,0.000049015245,0.00008130753,0.000116866875,0.6938518,0.00095405424,0.24051295,0.0009390777,0.062701955],"study_design_scores_gemma":[0.000011165217,0.000030967,0.00005793298,0.000009061611,0.0000060324383,0.000016431162,0.0000068831923,0.88644445,0.00048495794,0.11209917,0.00082487596,0.000008191575],"about_ca_topic_score_codex":0.002590043,"about_ca_topic_score_gemma":0.0024163877,"teacher_disagreement_score":0.004209871,"about_ca_system_score_codex":0.0011623738,"about_ca_system_score_gemma":0.0013301654,"threshold_uncertainty_score":0.014083445},"labels":[],"label_agreement":null},{"id":"W3211741236","doi":"","title":"Learning in two-player zero-sum partially observable Markov games with perfect recall","year":2021,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Observable; Zero (linguistics); Computer science; Markov chain; Recall; Zero-sum game; Markov process; Mathematics; Mathematical economics; Game theory; Statistics; Machine learning; Cognitive psychology; Psychology; Physics","score_opus":0.018763851378186695,"score_gpt":0.2540259796019725,"score_spread":0.23526212822378578,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3211741236","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.334663,0.00072297297,0.64598787,0.0027010797,0.00015296049,0.00020707854,0.00049016677,0.00044513852,0.014629784],"genre_scores_gemma":[0.98092777,0.00019161537,0.011028561,0.00017770335,0.00004558053,0.00012641956,0.00011546473,0.000028217377,0.007358606],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99774027,0.0008736433,0.000119061566,0.00043660944,0.00028590544,0.0005444482],"domain_scores_gemma":[0.98222256,0.0148248235,0.0011457276,0.00046536556,0.0005721671,0.0007693397],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034634813,0.0015134794,0.0037792262,0.00076652213,0.00086305256,0.00294524,0.0031840329,0.0029261461,0.00517834],"category_scores_gemma":[0.0155396415,0.0012197439,0.0009729419,0.00068957504,0.0036287215,0.0044038473,0.0032034363,0.0027904317,0.00042479267],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00075900217,0.00019563454,0.0011347239,0.00022845376,0.00013209324,0.0002879717,0.0002303545,0.82120717,0.000833207,0.16227426,0.0012393998,0.011477823],"study_design_scores_gemma":[0.00011321734,0.00007414071,0.00013872395,0.000015758214,0.000020045321,0.00002269343,0.000024973155,0.92719144,0.00022537034,0.072009206,0.00014291187,0.000021503665],"about_ca_topic_score_codex":0.0085022375,"about_ca_topic_score_gemma":0.006892487,"teacher_disagreement_score":0.0085022375,"about_ca_system_score_codex":0.002296489,"about_ca_system_score_gemma":0.002432427,"threshold_uncertainty_score":0.018316865},"labels":[],"label_agreement":null},{"id":"W3211776917","doi":"","title":"Learning Tree Interpretation from Object Representation for Deep Reinforcement Learning","year":2021,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Simon Fraser University","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Interpretation (philosophy); Representation (politics); Tree (set theory); Object (grammar); Machine learning; Programming language; Mathematics","score_opus":0.021558708445137575,"score_gpt":0.2752711759624968,"score_spread":0.2537124675173592,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3211776917","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011818546,0.00017319665,0.9861356,0.0001641201,0.000052643114,0.000021915717,0.00010871444,0.00065867114,0.0008665062],"genre_scores_gemma":[0.689185,0.0003352169,0.30583453,0.00021193919,0.00006441436,0.0001590811,0.0006378865,0.0003354582,0.003236381],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997844,0.000061995364,0.00001309914,0.00006180545,0.00005043279,0.000028298244],"domain_scores_gemma":[0.9991818,0.00042713605,0.00007385484,0.00011412024,0.00015513485,0.00004795562],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00068902003,0.00063395867,0.000987099,0.00057091453,0.0002532614,0.0009020405,0.0011480581,0.0014154318,0.0033428527],"category_scores_gemma":[0.003271149,0.0004216012,0.00064653717,0.0005993787,0.00068526046,0.0018547676,0.0010542851,0.0020056702,0.00061557867],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021642253,0.000086902815,0.00083560596,0.0001434177,0.000060298786,0.00011429022,0.00010049323,0.68513846,0.007658031,0.07430694,0.004598367,0.2267407],"study_design_scores_gemma":[0.0000049558284,0.000010832372,0.00003296142,0.0000058206774,0.000003075143,0.000004780659,0.0000029839437,0.97297716,0.00040300997,0.026308075,0.00024371201,0.0000026779342],"about_ca_topic_score_codex":0.0029559769,"about_ca_topic_score_gemma":0.0036311513,"teacher_disagreement_score":0.0033428527,"about_ca_system_score_codex":0.00085329247,"about_ca_system_score_gemma":0.0006928886,"threshold_uncertainty_score":0.011183023},"labels":[],"label_agreement":null},{"id":"W3212744543","doi":"10.48550/arxiv.2111.08172","title":"Off-Policy Actor-Critic with Emphatic Weightings","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Counterexample; Variety (cybernetics); Computer science; Variance (accounting); Work (physics); Gradient method; Mathematical optimization; Mathematics; Mathematical economics; Applied mathematics; Economics; Artificial intelligence; Discrete mathematics","score_opus":0.04648557504175391,"score_gpt":0.186150771689604,"score_spread":0.1396651966478501,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3212744543","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013934117,0.0001239407,0.9811383,0.00021635398,0.00003981972,0.00005326349,0.000019617859,0.00040651922,0.004067998],"genre_scores_gemma":[0.7898219,0.00017856441,0.20314634,0.00032636232,0.000042174594,0.00015854427,0.00007614978,0.00017121973,0.0060786814],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99900216,0.00041606172,0.000052031377,0.0001833266,0.00024141165,0.00010507139],"domain_scores_gemma":[0.99728334,0.0016964374,0.00024983572,0.000286552,0.00037487477,0.00010900825],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024298227,0.0011788674,0.000979299,0.0003685213,0.00032570184,0.0011571986,0.0012516803,0.0011076148,0.0025812846],"category_scores_gemma":[0.009372066,0.00056754093,0.00039015673,0.00034596308,0.001309658,0.0014460704,0.0013559684,0.0020239449,0.0006504048],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015066567,0.00007241113,0.00076761004,0.00010764181,0.00004008124,0.00008127973,0.00010003482,0.88782203,0.0025457453,0.044204045,0.0014675307,0.062640846],"study_design_scores_gemma":[0.000011389482,0.00001946194,0.000051132407,0.000007882728,0.0000042856836,0.000012441063,0.000006096624,0.9917128,0.00062874204,0.007111598,0.00043031797,0.0000037904686],"about_ca_topic_score_codex":0.002352051,"about_ca_topic_score_gemma":0.0022390713,"teacher_disagreement_score":0.0025812846,"about_ca_system_score_codex":0.0009329986,"about_ca_system_score_gemma":0.0014871244,"threshold_uncertainty_score":0.012850285},"labels":[],"label_agreement":null},{"id":"W3213430327","doi":"","title":"Pretraining Representations for Data-Efficient Reinforcement Learning","year":2021,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"HEC Montréal; Université de Montréal","funders":"","keywords":"Computer science; Reinforcement learning; Task (project management); Encoder; Artificial intelligence; Representation (politics); Key (lock); Code (set theory); Machine learning; Feature learning; External Data Representation","score_opus":0.15126924877798947,"score_gpt":0.2391395437057067,"score_spread":0.08787029492771722,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3213430327","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020607099,0.00018809998,0.974338,0.0003722833,0.000058014703,0.00010768923,0.00021751957,0.0025421195,0.0015691669],"genre_scores_gemma":[0.64813185,0.0001808145,0.34612492,0.0004476508,0.000059508024,0.00057720265,0.0011055811,0.00044755434,0.002924913],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991135,0.00031024305,0.000046739602,0.00024979154,0.0001646744,0.00011508106],"domain_scores_gemma":[0.9953754,0.0025850306,0.00028674747,0.0011224761,0.0004579466,0.00017235221],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020543467,0.0013992805,0.0013258582,0.0004260606,0.00046055694,0.0012115078,0.0024749576,0.0013861088,0.0044951416],"category_scores_gemma":[0.01309662,0.00084186223,0.00066430867,0.0005152151,0.0015191792,0.0026833417,0.0024541728,0.005069511,0.0017681429],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020977376,0.0003056328,0.0016779285,0.00019615464,0.00007621189,0.00009185643,0.00013798979,0.80238134,0.0062265275,0.016078122,0.0060840505,0.16653436],"study_design_scores_gemma":[0.000019585756,0.00004125723,0.000113482434,0.000015297235,0.0000054831658,0.000015509457,0.0000144233945,0.9830787,0.00209212,0.013763038,0.000833991,0.0000071264103],"about_ca_topic_score_codex":0.004254175,"about_ca_topic_score_gemma":0.006645767,"teacher_disagreement_score":0.0044951416,"about_ca_system_score_codex":0.0012023007,"about_ca_system_score_gemma":0.00200453,"threshold_uncertainty_score":0.015037775},"labels":[],"label_agreement":null},{"id":"W3213853783","doi":"","title":"On the Convergence and Sample Efficiency of Variance-Reduced Policy Gradient Method","year":2021,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Convexity; Truncation (statistics); Variance (accounting); Convergence (economics); Mathematics; Term (time); Variance reduction; Reinforcement learning; Applied mathematics; Mathematical optimization; Function (biology); Distribution (mathematics); Computer science; Combinatorics; Statistics; Mathematical analysis; Physics; Economics; Artificial intelligence","score_opus":0.05728126830244275,"score_gpt":0.21214144427449538,"score_spread":0.15486017597205265,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3213853783","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016800584,0.0006447174,0.97729105,0.0005521464,0.00007378034,0.000109133485,0.000041352698,0.00046548943,0.004021704],"genre_scores_gemma":[0.6171982,0.0009185199,0.3737909,0.00066602556,0.00016079636,0.00069636386,0.00030397985,0.00058894523,0.0056763897],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9974776,0.0013114457,0.00009881889,0.00026181716,0.00062462135,0.00022569491],"domain_scores_gemma":[0.9811746,0.015499248,0.00053664466,0.00096304173,0.001464838,0.0003616027],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0065692076,0.0012598075,0.001984282,0.0010394844,0.0006455477,0.001203361,0.0020100207,0.0016170994,0.003689339],"category_scores_gemma":[0.0400425,0.00067432324,0.0009044567,0.0005668838,0.0023641966,0.0021127076,0.002406326,0.0034282515,0.0007827448],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039412707,0.00020294658,0.0019537592,0.00024200436,0.000099389465,0.00013585007,0.00015224819,0.8437232,0.0024904278,0.09355958,0.002550018,0.05449659],"study_design_scores_gemma":[0.000018000848,0.000035966954,0.0000896407,0.000016771726,0.0000066014168,0.00001351551,0.0000067625097,0.99007344,0.00034190557,0.009145414,0.00024703573,0.000004920393],"about_ca_topic_score_codex":0.0046079494,"about_ca_topic_score_gemma":0.0035303663,"teacher_disagreement_score":0.0065692076,"about_ca_system_score_codex":0.0014362474,"about_ca_system_score_gemma":0.0028090084,"threshold_uncertainty_score":0.0347417},"labels":[],"label_agreement":null},{"id":"W3214099510","doi":"","title":"Brick-by-Brick: Combinatorial Construction with Deep Reinforcement Learning","year":2021,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Reinforcement learning; Construct (python library); Brick; Computer science; Object (grammar); Action (physics); Combinatorial explosion; Artificial intelligence; Space (punctuation); Theoretical computer science; Mathematics; Engineering; Programming language; Civil engineering","score_opus":0.020863170148706995,"score_gpt":0.15937401891041583,"score_spread":0.13851084876170883,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3214099510","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01750029,0.0002293278,0.97854006,0.00023239954,0.000047207584,0.000056189278,0.000047261907,0.0007743121,0.002572896],"genre_scores_gemma":[0.7535731,0.0002695549,0.24035797,0.00030796323,0.000048114918,0.00029033897,0.00023276763,0.00022338427,0.0046968637],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996327,0.00011481059,0.000016260947,0.00010391124,0.00007800299,0.000054340835],"domain_scores_gemma":[0.9989875,0.00055232993,0.00013229511,0.00013505743,0.00010148197,0.00009135472],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010531698,0.0011023307,0.0012722918,0.00039681702,0.0003563036,0.00086104183,0.0022481778,0.00138461,0.0031866203],"category_scores_gemma":[0.00294776,0.00070544565,0.00066094525,0.0004124748,0.0015162245,0.0015442623,0.0016187994,0.0022853524,0.0004470013],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000673542,0.00006539272,0.0005933122,0.000060132956,0.00004075689,0.000087928995,0.000041834977,0.94431907,0.0013852422,0.015641635,0.0013231423,0.036374174],"study_design_scores_gemma":[0.000006510679,0.000013749538,0.00002363547,0.000003778218,0.0000035126468,0.0000064688907,0.000001861387,0.9947178,0.00021848576,0.0048010475,0.00020079596,0.000002533261],"about_ca_topic_score_codex":0.0041368017,"about_ca_topic_score_gemma":0.0050379946,"teacher_disagreement_score":0.0041368017,"about_ca_system_score_codex":0.0011258862,"about_ca_system_score_gemma":0.0012588444,"threshold_uncertainty_score":0.010660291},"labels":[],"label_agreement":null},{"id":"W3214290601","doi":"","title":"Risk-Aware Transfer in Reinforcement Learning using Successor Features","year":2021,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Successor cardinal; Reinforcement learning; Computer science; Generalization; Variance (accounting); Task (project management); Representation (politics); Artificial intelligence; Function (biology); Domain (mathematical analysis); Machine learning; Bellman equation; Robot; Transfer of learning; Risk aversion (psychology); Expected utility hypothesis; Mathematical optimization; Mathematics; Engineering; Economics","score_opus":0.048609391666622925,"score_gpt":0.19407695932822996,"score_spread":0.14546756766160704,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3214290601","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0769974,0.00018343612,0.9206902,0.00019732463,0.00002466286,0.000050960964,0.000028025739,0.00032715997,0.0015008954],"genre_scores_gemma":[0.9570548,0.00007565894,0.041706126,0.00006244193,0.000019687805,0.000084832805,0.000030288104,0.000036346726,0.0009297542],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989605,0.00040901045,0.00005923624,0.00020896606,0.00024829182,0.00011399022],"domain_scores_gemma":[0.99644357,0.0023731932,0.00037809886,0.0003332011,0.00029747226,0.0001745282],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028040707,0.0007915259,0.0011607707,0.00045847043,0.0002755028,0.00091547635,0.0010361015,0.00088034704,0.001199805],"category_scores_gemma":[0.0098974,0.00040487206,0.00061999046,0.00031955252,0.0012861106,0.0019787136,0.0014330695,0.0016236532,0.00018356531],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015222139,0.00012247254,0.0015514377,0.000062930405,0.000056906425,0.00008483862,0.00010123994,0.911739,0.0024701785,0.02266725,0.00036616463,0.0606254],"study_design_scores_gemma":[0.000011176168,0.00007319413,0.0001732443,0.00000611247,0.000007347216,0.000014038699,0.0000039591437,0.98721445,0.0004515477,0.011919019,0.00011951594,0.000006404525],"about_ca_topic_score_codex":0.0011293606,"about_ca_topic_score_gemma":0.0008464419,"teacher_disagreement_score":0.0028040707,"about_ca_system_score_codex":0.00091569184,"about_ca_system_score_gemma":0.00088557997,"threshold_uncertainty_score":0.014829516},"labels":[],"label_agreement":null},{"id":"W3214824629","doi":"10.1109/icra46639.2022.9812407","title":"Learning Interactive Driving Policies via Data-driven Simulation","year":2022,"lang":"en","type":"article","venue":"2022 International Conference on Robotics and Automation (ICRA)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Bottleneck; Domain (mathematical analysis); Policy learning; Enhanced Data Rates for GSM Evolution; Transfer of learning; State (computer science); Machine learning; Artificial intelligence","score_opus":0.04795188783829782,"score_gpt":0.31789167568736826,"score_spread":0.26993978784907047,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3214824629","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08101098,0.00009683618,0.91413414,0.00026319313,0.000044867193,0.00010367804,0.00017809166,0.0015812415,0.0025869247],"genre_scores_gemma":[0.9275121,0.00007681839,0.07069095,0.00007645004,0.000014648161,0.00017493848,0.0002757118,0.000116977986,0.0010615034],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997645,0.00006631802,0.00001225067,0.000060068378,0.00005967753,0.000037096026],"domain_scores_gemma":[0.99859685,0.00086808583,0.00011569703,0.00015983543,0.00015658616,0.000102939775],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006974988,0.0007000837,0.00069057994,0.00038616374,0.00031345713,0.00076327886,0.0013039489,0.0009862784,0.0016826285],"category_scores_gemma":[0.0033819636,0.0006306743,0.0005537717,0.00026507306,0.0009059828,0.0008993243,0.0013587698,0.0013195656,0.00030018733],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000019282941,0.00001458464,0.0002324458,0.000009510392,0.000006547047,0.000012790116,0.000010630768,0.9951332,0.00041733723,0.00088445673,0.00011371194,0.0031455809],"study_design_scores_gemma":[0.0000027786566,0.000003855727,0.000015805777,7.2157087e-7,5.980794e-7,0.0000014899356,0.0000014994246,0.99920017,0.00019541271,0.00048698837,0.00008974272,9.601398e-7],"about_ca_topic_score_codex":0.0060294424,"about_ca_topic_score_gemma":0.005171039,"teacher_disagreement_score":0.0060294424,"about_ca_system_score_codex":0.00086265715,"about_ca_system_score_gemma":0.001242818,"threshold_uncertainty_score":0.011988699},"labels":[],"label_agreement":null},{"id":"W3215048956","doi":"10.1109/icra46639.2022.9812276","title":"VISTA 2.0: An Open, Data-driven Simulator for Multimodal Sensing and Policy Learning for Autonomous Vehicles","year":2022,"lang":"en","type":"article","venue":"2022 International Conference on Robotics and Automation (ICRA)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":71,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toyota Motor Corporation (Canada); University of Toronto","funders":"National Science Foundation","keywords":"Computer science; Robustness (evolution); Software deployment; Key (lock); Testbed; Artificial intelligence; Viewpoints; World Wide Web; Software engineering","score_opus":0.07474863535135583,"score_gpt":0.34999590304230355,"score_spread":0.2752472676909477,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3215048956","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.048302963,0.00023218383,0.8196457,0.00081421493,0.00051387755,0.00082505017,0.015226804,0.088042445,0.026396671],"genre_scores_gemma":[0.41033044,0.0006381966,0.5254735,0.00055411697,0.000081794555,0.0027535975,0.031164125,0.015339559,0.013664688],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997204,0.000053829477,0.000025391117,0.00004631297,0.00011630516,0.00003769781],"domain_scores_gemma":[0.99922776,0.0003857062,0.00004603999,0.00009215276,0.00014217592,0.00010599538],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007465336,0.0010822058,0.0006387731,0.00068579585,0.00045443632,0.0009918209,0.0028172305,0.001373159,0.013879102],"category_scores_gemma":[0.0028310178,0.00086427503,0.001248481,0.00050766027,0.0005837398,0.001082425,0.0015243269,0.0020433948,0.0022370492],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027243161,0.00020219272,0.0027288836,0.00025257826,0.00013493276,0.0001525362,0.0002067057,0.9199073,0.0052700453,0.012363319,0.031514604,0.026994497],"study_design_scores_gemma":[0.000058736718,0.000022585926,0.00012941002,0.00001127433,0.000008630729,0.0000171973,0.000013196096,0.980785,0.0021896334,0.0023449215,0.014403054,0.000016452104],"about_ca_topic_score_codex":0.016491368,"about_ca_topic_score_gemma":0.020304097,"teacher_disagreement_score":0.016491368,"about_ca_system_score_codex":0.0011045965,"about_ca_system_score_gemma":0.0025798818,"threshold_uncertainty_score":0.04643017},"labels":[],"label_agreement":null},{"id":"W359568995","doi":"","title":"A convergent O ( n ) algorithm for off-policy temporal-difference learning with linear function approximation","year":2008,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":120,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Temporal difference learning; Markov decision process; Algorithm; Stochastic approximation; Function (biology); Reinforcement learning; Stochastic gradient descent; Markov process; Mathematics; Quadratic equation; Approximation algorithm; Norm (philosophy); Computer science; Function approximation; Mathematical optimization; Applied mathematics; Artificial intelligence; Statistics; Artificial neural network","score_opus":0.02640782854945645,"score_gpt":0.2470170740798348,"score_spread":0.22060924553037836,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W359568995","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0015099886,0.00009158822,0.9967186,0.000102976985,0.00004331573,0.00004912211,0.000015595566,0.0004685019,0.001000389],"genre_scores_gemma":[0.12206259,0.00014114665,0.87153715,0.00031125717,0.00007819555,0.00042556677,0.00020744029,0.0002847689,0.0049518463],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99889755,0.00021881312,0.000071139235,0.00027966185,0.00041197974,0.0001208145],"domain_scores_gemma":[0.99804604,0.0010505291,0.00013083704,0.00023183982,0.0004217121,0.00011895389],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001867837,0.001108586,0.0012206313,0.0006826366,0.00062339124,0.0010510079,0.0025793433,0.0017224719,0.008051493],"category_scores_gemma":[0.008128007,0.000618944,0.0007617791,0.0006622829,0.0011876447,0.0017997451,0.0024313615,0.0033453386,0.0030954794],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032744437,0.00019601775,0.0007341583,0.00015851657,0.000049836803,0.00012633778,0.00016384613,0.4185304,0.0053052898,0.0661085,0.006849732,0.5014498],"study_design_scores_gemma":[0.000024600533,0.00003043525,0.000038364284,0.0000086241425,0.0000035122234,0.000027387588,0.0000067360606,0.9873511,0.00076856447,0.010486018,0.0012480767,0.000006507244],"about_ca_topic_score_codex":0.004988086,"about_ca_topic_score_gemma":0.004152969,"teacher_disagreement_score":0.008051493,"about_ca_system_score_codex":0.0017154083,"about_ca_system_score_gemma":0.0022526467,"threshold_uncertainty_score":0.026934922},"labels":[],"label_agreement":null},{"id":"W36013382","doi":"10.3390/life12081203","title":"A decision-theoretic approach to task assistance for persons with dementia","year":2005,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":160,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Rehabilitation Institute; University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Partially observable Markov decision process; Dementia; Task (project management); Personalization; Computer science; Independence (probability theory); Markov decision process; Process (computing); Cognition; Markov process; Artificial intelligence; Human–computer interaction; Cognitive psychology; Risk analysis (engineering); Machine learning; Psychology; Disease; Markov model; Markov chain; Medicine; Engineering; Neuroscience; Mathematics","score_opus":0.013658746007665902,"score_gpt":0.24319007784262545,"score_spread":0.22953133183495955,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W36013382","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.057715654,0.00074231334,0.9090068,0.0055390596,0.00015883805,0.00037963514,0.0002035982,0.0003024825,0.025951698],"genre_scores_gemma":[0.8420443,0.0006342275,0.14843246,0.00045367941,0.00014965318,0.0009457458,0.00019765868,0.000047813654,0.0070943856],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987937,0.0007371953,0.00005671373,0.00016950787,0.00011992931,0.00012302383],"domain_scores_gemma":[0.99564064,0.0035743213,0.00015646617,0.00008509336,0.00027623176,0.0002672881],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027678707,0.0010826122,0.0009786194,0.0007107405,0.0007727061,0.001376983,0.0016560378,0.0018895612,0.009602443],"category_scores_gemma":[0.007836957,0.00043285897,0.0007551829,0.00046764867,0.0013438859,0.0014069458,0.0019873362,0.0016529976,0.0007599468],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043248676,0.000445336,0.0020284934,0.00023173357,0.00011384945,0.00022790798,0.00048441737,0.83856595,0.00042292193,0.06572887,0.0042195483,0.087098464],"study_design_scores_gemma":[0.00008404198,0.00013644966,0.00023201064,0.000037581995,0.000020193866,0.000040427363,0.00012346529,0.9018704,0.00015275457,0.09579985,0.0014855671,0.000017136823],"about_ca_topic_score_codex":0.0055846213,"about_ca_topic_score_gemma":0.004793356,"teacher_disagreement_score":0.009602443,"about_ca_system_score_codex":0.001734634,"about_ca_system_score_gemma":0.0027996406,"threshold_uncertainty_score":0.032123446},"labels":[],"label_agreement":null},{"id":"W409933044","doi":"10.1609/aiide.v8i1.12515","title":"Statechart-Based AI in Practice","year":2012,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Modular design; USable; Reuse; Computer science; Tree (set theory); Artificial intelligence; Software engineering; Point (geometry); Scale (ratio); Human–computer interaction; Programming language; Engineering; World Wide Web","score_opus":0.039331803373794744,"score_gpt":0.30429367141374575,"score_spread":0.26496186803995103,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W409933044","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0019501484,0.00040529063,0.96872026,0.0012103776,0.00012017735,0.0000881104,0.0001012802,0.0025469747,0.024857458],"genre_scores_gemma":[0.16447107,0.0015574759,0.81567866,0.0005954661,0.00010421947,0.00042099415,0.00053432444,0.00076249137,0.015875315],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977245,0.0008877287,0.00019529353,0.00033968835,0.00069738587,0.0001554197],"domain_scores_gemma":[0.99593806,0.0019945626,0.00016428962,0.0010584453,0.0006590449,0.00018558011],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036588816,0.00066342636,0.00036925578,0.0010422688,0.00074227125,0.004050668,0.0018709903,0.0017624233,0.020184197],"category_scores_gemma":[0.009783663,0.00072305684,0.0009133022,0.0008259108,0.0035971678,0.005555217,0.0028542194,0.0029710496,0.0044212905],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000036072703,0.00004814506,0.0003632537,0.00032393777,0.000029273895,0.00013297219,0.00067702186,0.03406035,0.0030454523,0.8884145,0.006719102,0.06614979],"study_design_scores_gemma":[0.00004140402,0.00005624757,0.00012439821,0.00023559021,0.000029832772,0.00017607944,0.00016595726,0.1413895,0.0036321809,0.6680929,0.18602645,0.000029407533],"about_ca_topic_score_codex":0.004368934,"about_ca_topic_score_gemma":0.0042516687,"teacher_disagreement_score":0.020184197,"about_ca_system_score_codex":0.001841175,"about_ca_system_score_gemma":0.0021556222,"threshold_uncertainty_score":0.067522824},"labels":[],"label_agreement":null},{"id":"W41346994","doi":"","title":"Bootstrapping the Learning Process for the Semi-automated Design of a Challenging Game AI","year":2004,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Reinforcement learning; Bootstrapping (finance); Artificial intelligence; Process (computing); Key (lock); Game design; Representation (politics); Decomposition; Control (management); Machine learning; Human–computer interaction; Programming language","score_opus":0.03322517527403423,"score_gpt":0.2744859141496011,"score_spread":0.24126073887556687,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W41346994","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06559717,0.00010476325,0.9311274,0.00029958654,0.00001994028,0.000118805976,0.000029420287,0.00032010415,0.0023827343],"genre_scores_gemma":[0.9468264,0.000043906548,0.051612522,0.00005889078,0.000014961306,0.00018821844,0.000032592252,0.000050964314,0.0011715099],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985084,0.00085053965,0.000051920673,0.00022194066,0.0001932219,0.0001740443],"domain_scores_gemma":[0.9846971,0.012912577,0.0007651627,0.00048475782,0.00063289824,0.0005074781],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038193238,0.0011193644,0.0016482327,0.0005982826,0.0006037819,0.0010550438,0.0016288495,0.001781885,0.003731481],"category_scores_gemma":[0.019312203,0.00087839185,0.0006547091,0.00029311198,0.002165993,0.0012552011,0.0022912866,0.0025653252,0.0003880966],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015857864,0.000055768785,0.00036935485,0.00006860476,0.000025199512,0.000039308205,0.00012291507,0.9736376,0.00086488406,0.010817663,0.0002403503,0.013599855],"study_design_scores_gemma":[0.00000949552,0.000023565963,0.000028148013,0.0000036271645,0.0000022014071,0.0000023307155,0.0000036658173,0.9954711,0.00011295427,0.004295016,0.000046029294,0.0000019551303],"about_ca_topic_score_codex":0.0048704506,"about_ca_topic_score_gemma":0.004070472,"teacher_disagreement_score":0.0048704506,"about_ca_system_score_codex":0.0013108829,"about_ca_system_score_gemma":0.001850433,"threshold_uncertainty_score":0.020198703},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"simulation_or_modeling","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"simulation_or_modeling","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"low"}],"label_agreement":"agree"},{"id":"W4200140111","doi":"10.1109/iros51168.2021.9636140","title":"Memory-based Deep Reinforcement Learning for POMDPs","year":2021,"lang":"en","type":"article","venue":"2021 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":77,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Compute Canada","keywords":"Reinforcement learning; Markov decision process; Computer science; Observable; Artificial intelligence; Partially observable Markov decision process; Noise (video); Robotics; Component (thermodynamics); Sensitivity (control systems); Feature (linguistics); Deep learning; Markov process; Markov chain; Machine learning; Robot; Markov model","score_opus":0.06087323939367912,"score_gpt":0.30000014092519434,"score_spread":0.23912690153151522,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4200140111","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0355684,0.00042117687,0.9604839,0.00027831268,0.000060675007,0.000050039438,0.00010474956,0.0010343855,0.001998362],"genre_scores_gemma":[0.9218271,0.00015889114,0.076022945,0.00015686249,0.00002236306,0.00013178293,0.00015984438,0.0000693151,0.001450848],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996376,0.00009724244,0.000025467194,0.000082331666,0.000084002124,0.000073258176],"domain_scores_gemma":[0.99836856,0.0010971255,0.00015180415,0.00009766298,0.00019960561,0.00008525095],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010855757,0.0009095628,0.0011618591,0.00035559217,0.00034679275,0.0006421455,0.0013223012,0.00087961834,0.0024530678],"category_scores_gemma":[0.0037351968,0.0005055127,0.0004843209,0.0003495932,0.00082275626,0.00096250133,0.001192551,0.0018087202,0.00027124616],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005079453,0.00003341674,0.00043262113,0.000051893072,0.000019725901,0.000032336855,0.000028836277,0.9695182,0.00048808582,0.003692927,0.00043962232,0.02521153],"study_design_scores_gemma":[0.00000389851,0.000008045963,0.000016675678,0.0000019406145,0.0000013712558,0.0000020593293,0.000001501262,0.9984201,0.000110249624,0.0013774023,0.000055509314,0.0000011104621],"about_ca_topic_score_codex":0.009951817,"about_ca_topic_score_gemma":0.0095427595,"teacher_disagreement_score":0.009951817,"about_ca_system_score_codex":0.0013511986,"about_ca_system_score_gemma":0.0017001939,"threshold_uncertainty_score":0.019787788},"labels":[],"label_agreement":null},{"id":"W4200372371","doi":"10.1002/cpe.6743","title":"Scalable grid‐based approximation algorithms for partially observable Markov decision processes","year":2021,"lang":"en","type":"article","venue":"Concurrency and Computation Practice and Experience","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Partially observable Markov decision process; Computer science; Markov decision process; Scalability; Observable; Mathematical optimization; Implementation; Grid; Markov process; Process (computing); Focus (optics); Algorithm; Markov chain; Markov model; Machine learning; Mathematics","score_opus":0.04756454524991755,"score_gpt":0.3385360643285482,"score_spread":0.29097151907863067,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4200372371","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035366956,0.00046718004,0.9598309,0.000335883,0.000053212927,0.000069272995,0.00012671722,0.0008190753,0.002930943],"genre_scores_gemma":[0.81634337,0.00026606067,0.18128155,0.00009671589,0.000028017006,0.00019961415,0.00023917676,0.000102057405,0.0014433507],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99933416,0.00025652355,0.000039265626,0.000112899914,0.00015387268,0.00010310739],"domain_scores_gemma":[0.9962585,0.0028295857,0.00027497634,0.0002623467,0.00023111535,0.00014356946],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016516919,0.000707464,0.0014915925,0.0005652175,0.0006141212,0.0010020459,0.001607456,0.0008401305,0.0036099073],"category_scores_gemma":[0.0053429543,0.0005073301,0.00062056805,0.000857225,0.00091310334,0.0011467645,0.0016591634,0.0015651203,0.00035387216],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000044741744,0.000022041158,0.00036634525,0.000027252003,0.000013291677,0.000017375834,0.000023127062,0.98259246,0.00012090229,0.007376273,0.00034301955,0.0090531865],"study_design_scores_gemma":[0.0000058050946,0.0000030637823,0.000014247415,0.000001827372,8.4317276e-7,0.0000015390481,0.000002931164,0.9970799,0.000024029161,0.002811068,0.00005406598,6.5372114e-7],"about_ca_topic_score_codex":0.0151762515,"about_ca_topic_score_gemma":0.0131347515,"teacher_disagreement_score":0.0151762515,"about_ca_system_score_codex":0.0016961447,"about_ca_system_score_gemma":0.0021152876,"threshold_uncertainty_score":0.030175805},"labels":[],"label_agreement":null},{"id":"W4200400844","doi":"10.32920/17313137","title":"Safe Driving Of Autonomous Vehicles Through Improved Deep Reinforcement Learning","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Obstacle; Train; Obstacle avoidance; Perception; Trajectory; Action (physics); Object (grammar); Robot; Mobile robot","score_opus":0.0186631170123467,"score_gpt":0.2561486766434662,"score_spread":0.2374855596311195,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4200400844","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.050587077,0.00019784771,0.9456073,0.00022743136,0.00004287862,0.000035043333,0.000023630035,0.00064399594,0.0026346669],"genre_scores_gemma":[0.94533384,0.000072366376,0.052103464,0.000090717585,0.00001535055,0.000050880528,0.000044797118,0.00004977617,0.0022388704],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997404,0.0000605381,0.000010878598,0.000068468325,0.00007113146,0.000048525348],"domain_scores_gemma":[0.9994548,0.00023313198,0.00008662238,0.000046756548,0.000121816614,0.000056886194],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007411675,0.0006648758,0.0005947437,0.0002581644,0.00026054864,0.00057949463,0.0010519872,0.00067269255,0.0010359302],"category_scores_gemma":[0.0018680692,0.00043287777,0.00039640962,0.00014282709,0.0007235366,0.0007331168,0.0010080338,0.0010767884,0.00021768427],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002818694,0.000031428768,0.0006258788,0.000017657687,0.000014821721,0.00004284515,0.000035348075,0.97786075,0.0016242424,0.003383567,0.00029267647,0.016042503],"study_design_scores_gemma":[0.0000018365077,0.000009157711,0.000023210378,0.0000010527386,9.879167e-7,0.0000025495806,0.0000012267285,0.99921143,0.0001282872,0.00055493426,0.00006438669,9.924796e-7],"about_ca_topic_score_codex":0.005861239,"about_ca_topic_score_gemma":0.0049833073,"teacher_disagreement_score":0.005861239,"about_ca_system_score_codex":0.0007834156,"about_ca_system_score_gemma":0.0010703758,"threshold_uncertainty_score":0.011654258},"labels":[],"label_agreement":null},{"id":"W4200507047","doi":"10.32920/17313137.v1","title":"Safe Driving Of Autonomous Vehicles Through Improved Deep Reinforcement Learning","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Obstacle; Train; Obstacle avoidance; Perception; Trajectory; Action (physics); Object (grammar); Robot; Mobile robot","score_opus":0.0186631170123467,"score_gpt":0.2561486766434662,"score_spread":0.2374855596311195,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4200507047","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.050587077,0.00019784771,0.9456073,0.00022743136,0.00004287862,0.000035043333,0.000023630035,0.00064399594,0.0026346669],"genre_scores_gemma":[0.94533384,0.000072366376,0.052103464,0.000090717585,0.00001535055,0.000050880528,0.000044797118,0.00004977617,0.0022388704],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997404,0.0000605381,0.000010878598,0.000068468325,0.00007113146,0.000048525348],"domain_scores_gemma":[0.9994548,0.00023313198,0.00008662238,0.000046756548,0.000121816614,0.000056886194],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007411675,0.0006648758,0.0005947437,0.0002581644,0.00026054864,0.00057949463,0.0010519872,0.00067269255,0.0010359302],"category_scores_gemma":[0.0018680692,0.00043287777,0.00039640962,0.00014282709,0.0007235366,0.0007331168,0.0010080338,0.0010767884,0.00021768427],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002818694,0.000031428768,0.0006258788,0.000017657687,0.000014821721,0.00004284515,0.000035348075,0.97786075,0.0016242424,0.003383567,0.00029267647,0.016042503],"study_design_scores_gemma":[0.0000018365077,0.000009157711,0.000023210378,0.0000010527386,9.879167e-7,0.0000025495806,0.0000012267285,0.99921143,0.0001282872,0.00055493426,0.00006438669,9.924796e-7],"about_ca_topic_score_codex":0.005861239,"about_ca_topic_score_gemma":0.0049833073,"teacher_disagreement_score":0.005861239,"about_ca_system_score_codex":0.0007834156,"about_ca_system_score_gemma":0.0010703758,"threshold_uncertainty_score":0.011654258},"labels":[],"label_agreement":null},{"id":"W4200555244","doi":"10.11606/t.55.2021.tde-21122021-111842","title":"Synthesizing interpretable strategies for real-time planning in zero-sum games","year":2021,"lang":"en","type":"dissertation","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Division of Mathematical Sciences; Centro de Ciências Matemáticas Aplicadas à Indústria; Universidade Federal de Viçosa; Fundação de Amparo à Pesquisa do Estado de São Paulo; Conselho Nacional de Desenvolvimento Científico e Tecnológico; Compute Canada; Coordenação de Aperfeiçoamento de Pessoal de Nível Superior; Canadian Institute for Advanced Research","keywords":"Computer science; Scripting language; Action (physics); Artificial intelligence; Set (abstract data type); Abstraction; Zero (linguistics); Machine learning; Programming language","score_opus":0.018196227617453244,"score_gpt":0.2910867091637649,"score_spread":0.2728904815463116,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4200555244","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04878865,0.0002432528,0.9389384,0.00057837943,0.00008999609,0.000327651,0.0002902113,0.002972971,0.0077704317],"genre_scores_gemma":[0.4210334,0.0003084183,0.5718736,0.00030711605,0.000026549633,0.0005653104,0.0008185817,0.0006972448,0.0043697637],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99884135,0.00036056992,0.0001031775,0.0002476096,0.00033765042,0.0001095487],"domain_scores_gemma":[0.9975068,0.0016821363,0.00018561559,0.00027092293,0.00023625766,0.000118213975],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014396994,0.0014562559,0.00070558576,0.0006102209,0.0005023937,0.0016356792,0.0013899601,0.00093910727,0.004040234],"category_scores_gemma":[0.007267307,0.00053111714,0.0014669982,0.00035673488,0.002319347,0.0017001883,0.0018851048,0.001837605,0.0007383779],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022259643,0.00020919177,0.0017262439,0.00045877823,0.00007658572,0.00046639552,0.001130514,0.73018426,0.010312209,0.14568761,0.0031435874,0.10638207],"study_design_scores_gemma":[0.000056780216,0.00007603019,0.00013370709,0.000050211933,0.000031539872,0.000045196797,0.00012364714,0.91106963,0.0064380807,0.07698185,0.0049748733,0.00001847028],"about_ca_topic_score_codex":0.003796235,"about_ca_topic_score_gemma":0.008800021,"teacher_disagreement_score":0.004040234,"about_ca_system_score_codex":0.0016078439,"about_ca_system_score_gemma":0.0020696684,"threshold_uncertainty_score":0.013515949},"labels":[],"label_agreement":null},{"id":"W4206522553","doi":"10.22215/etd/2021-14679","title":"Evolution of Multiobjective Neuromodulated Neurocontrollers for Multi-Robot Systems","year":2021,"lang":"en","type":"dissertation","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Queen's University","funders":"","keywords":"Neuroevolution; Artificial intelligence; Artificial neural network; Computer science; Inheritance (genetic algorithm); Genetic algorithm; Machine learning; Biology","score_opus":0.02769569525311363,"score_gpt":0.28359359275935947,"score_spread":0.25589789750624586,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4206522553","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28174722,0.00045555306,0.7039483,0.00038652937,0.000110575726,0.00008274137,0.000051423034,0.00025932962,0.012958306],"genre_scores_gemma":[0.9338007,0.00015730056,0.061620396,0.0000665961,0.000011495295,0.00010057197,0.000034996985,0.000030570798,0.0041774414],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998685,0.0000363285,0.000007380463,0.000028150154,0.000038985352,0.000020625539],"domain_scores_gemma":[0.99968266,0.0001233905,0.00005737672,0.000028117853,0.00007708455,0.00003141611],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004414774,0.00038761337,0.00032541444,0.0002242506,0.00023996789,0.00063101156,0.0006362907,0.00052596896,0.0014014256],"category_scores_gemma":[0.001317331,0.00020295542,0.00037908938,0.00014832617,0.00040405543,0.00036086515,0.00076761964,0.0006304573,0.00012496077],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002500522,0.000030915715,0.00043663706,0.000027905344,0.0000372903,0.00005205371,0.000039641636,0.9658138,0.007728532,0.008150834,0.00024249658,0.017414879],"study_design_scores_gemma":[0.000004762211,0.000026123438,0.00011778333,0.0000040379223,0.0000037103564,0.000010673914,0.0000065793574,0.99660933,0.00081539346,0.0021280397,0.00027083955,0.0000026802875],"about_ca_topic_score_codex":0.0013437727,"about_ca_topic_score_gemma":0.0013763432,"teacher_disagreement_score":0.0014014256,"about_ca_system_score_codex":0.00062328315,"about_ca_system_score_gemma":0.0004316982,"threshold_uncertainty_score":0.0046882033},"labels":[],"label_agreement":null},{"id":"W4206914807","doi":"10.1109/ssci50451.2021.9660081","title":"Investigation of Maximization Bias in Sarsa Variants","year":2021,"lang":"en","type":"article","venue":"2021 IEEE Symposium Series on Computational Intelligence (SSCI)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Maximization; Reinforcement learning; Computer science; Randomness; Artificial intelligence; Q-learning; Expectation–maximization algorithm; Variance (accounting); Machine learning; Statistics; Mathematical optimization; Mathematics; Maximum likelihood; Economics","score_opus":0.05639518078776527,"score_gpt":0.2753421580849772,"score_spread":0.2189469772972119,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4206914807","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6922896,0.0023723084,0.29226577,0.0013264517,0.00023769801,0.00027697254,0.00017204475,0.0012243473,0.009834868],"genre_scores_gemma":[0.9603121,0.0002261884,0.037944656,0.00021736171,0.000030519277,0.000113116585,0.000119550605,0.00015025653,0.00088612345],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9949655,0.0022520993,0.00035080593,0.00068361935,0.0011308403,0.00061715173],"domain_scores_gemma":[0.93372667,0.051867474,0.0034015805,0.0049615856,0.00486747,0.0011752185],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012069274,0.0011560319,0.0014947796,0.0010508873,0.00054889603,0.0016666909,0.0016319,0.0013255784,0.0019709687],"category_scores_gemma":[0.06358129,0.0005238141,0.0007694051,0.000556412,0.0013605063,0.0020858948,0.0015622177,0.002188388,0.00032159072],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008851227,0.0005947767,0.015065464,0.0003554651,0.00028240916,0.0001969737,0.00024741827,0.8947628,0.004998381,0.019665958,0.0022469214,0.060698245],"study_design_scores_gemma":[0.000064937085,0.00032216578,0.0011284568,0.00004534126,0.000030282674,0.00010149711,0.000061258674,0.9889703,0.002192625,0.006545586,0.0005223037,0.000015327309],"about_ca_topic_score_codex":0.0021667196,"about_ca_topic_score_gemma":0.0019411637,"teacher_disagreement_score":0.012069274,"about_ca_system_score_codex":0.0015380651,"about_ca_system_score_gemma":0.0031510065,"threshold_uncertainty_score":0.06382918},"labels":[],"label_agreement":null},{"id":"W4210257517","doi":"10.1109/cdc45484.2021.9682777","title":"Convergence and Near Optimality of Q-Learning with Finite Memory for Partially Observed Models","year":2021,"lang":"en","type":"article","venue":"2021 60th IEEE Conference on Decision and Control (CDC)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Partially observable Markov decision process; Reinforcement learning; Markov decision process; Convergence (economics); Computer science; Quantization (signal processing); Q-learning; Mathematical optimization; Limit (mathematics); State space; Optimal control; Markov process; Mathematics; Algorithm; Artificial intelligence","score_opus":0.06092869236888402,"score_gpt":0.2648753527444318,"score_spread":0.2039466603755478,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4210257517","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020581812,0.00035669009,0.97689396,0.00032414251,0.000025357298,0.000049989252,0.000030585634,0.00014833533,0.0015891377],"genre_scores_gemma":[0.82487506,0.00060889084,0.17081931,0.00028895738,0.00006500266,0.00034080408,0.00015120854,0.00014272812,0.0027080998],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976223,0.0011665547,0.00011950682,0.00039679868,0.0004755895,0.00021917786],"domain_scores_gemma":[0.9727553,0.02366278,0.000991724,0.0007027327,0.0014862487,0.00040115762],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00703588,0.0010168949,0.0019659689,0.0010187649,0.00083017914,0.0013456342,0.0016877411,0.001788868,0.001977296],"category_scores_gemma":[0.033418454,0.0007332667,0.00092573714,0.0006796508,0.0034742581,0.0025758974,0.0025882758,0.0028875414,0.00029112236],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000111792855,0.0000713256,0.00087791955,0.00011083764,0.000048785983,0.00005953546,0.00013949713,0.9113363,0.00063193904,0.06865873,0.00038776523,0.017565643],"study_design_scores_gemma":[0.000011009229,0.000029251574,0.000052771724,0.000011526432,0.0000033973947,0.0000079877445,0.0000065941476,0.97925365,0.0001933329,0.020327253,0.000098357006,0.0000048619877],"about_ca_topic_score_codex":0.0065704593,"about_ca_topic_score_gemma":0.0026974587,"teacher_disagreement_score":0.00703588,"about_ca_system_score_codex":0.0020598117,"about_ca_system_score_gemma":0.0031894753,"threshold_uncertainty_score":0.03720975},"labels":[],"label_agreement":null},{"id":"W4210266050","doi":"10.1109/cinti53070.2021.9668479","title":"Deep Reinforcement Learning with DQN vs. PPO in VizDoom","year":2021,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Convergence (economics); Reinforcement; Quality (philosophy); Track (disk drive); Object (grammar); Machine learning; Human–computer interaction; Engineering","score_opus":0.010729311580889345,"score_gpt":0.2251434012773191,"score_spread":0.21441408969642975,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4210266050","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.053113542,0.0007661901,0.9339629,0.00069276005,0.00023515717,0.000094422874,0.00008781176,0.001827124,0.009220084],"genre_scores_gemma":[0.9169317,0.00014880586,0.07850231,0.00028570573,0.000036068242,0.00008863674,0.00008188799,0.00009439535,0.003830404],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996866,0.00007415618,0.000017768143,0.000103810264,0.000060167808,0.0000574661],"domain_scores_gemma":[0.9993506,0.00029934756,0.000050403694,0.00009789899,0.000120838406,0.00008090933],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010946123,0.00050007435,0.000783718,0.00020864523,0.00029832427,0.00080479804,0.0011846167,0.0011898096,0.0032401155],"category_scores_gemma":[0.0028847326,0.0003170327,0.00029150565,0.00024265656,0.00086712424,0.0011293773,0.001429373,0.001683279,0.00047892798],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041181175,0.00015879363,0.0011946162,0.0000961018,0.000043063577,0.00007547298,0.00005844633,0.8553181,0.0027113482,0.017726887,0.0025059686,0.11969931],"study_design_scores_gemma":[0.000026666421,0.00005111081,0.000085574735,0.0000050905696,0.000005288768,0.0000096835765,0.000005248326,0.99431145,0.00062506075,0.004224716,0.0006458197,0.0000042547863],"about_ca_topic_score_codex":0.006710474,"about_ca_topic_score_gemma":0.0046066786,"teacher_disagreement_score":0.006710474,"about_ca_system_score_codex":0.00074839516,"about_ca_system_score_gemma":0.0010752903,"threshold_uncertainty_score":0.013342798},"labels":[],"label_agreement":null},{"id":"W4210732917","doi":"10.1016/bs.hna.2021.12.016","title":"Machine learning and control theory","year":2022,"lang":"en","type":"book-chapter","venue":"Handbook of numerical analysis","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Chinese University of Hong Kong; Saint Mary’s University; National Science Foundation","keywords":"Computer science; Artificial intelligence; Machine learning; Online machine learning; Algorithmic learning theory; Reinforcement learning; Computational learning theory; Control (management); Stochastic gradient descent; Probably approximately correct learning; Field (mathematics); Active learning (machine learning); Artificial neural network; Mathematics","score_opus":0.009157608706874624,"score_gpt":0.21918081536263434,"score_spread":0.21002320665575971,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4210732917","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0009472438,0.27970758,0.2853643,0.0046732975,0.0048882687,0.00013254432,0.0015988399,0.001955316,0.42073262],"genre_scores_gemma":[0.03556735,0.19310264,0.20336072,0.0038820438,0.0062291254,0.000661237,0.0029622465,0.0016051534,0.5526295],"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99953246,0.000091144124,0.000023336665,0.00008565938,0.0002453844,0.000022087821],"domain_scores_gemma":[0.99935097,0.00039682435,0.000025525434,0.00007880472,0.00012259863,0.000025375923],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00056694297,0.0016840196,0.0023965999,0.0018675686,0.0005689662,0.0025316146,0.0015901581,0.0015196688,0.038294394],"category_scores_gemma":[0.0017285445,0.00074225035,0.00048300048,0.003136225,0.0016517822,0.0024251034,0.0010068599,0.0038755971,0.024538632],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001814186,0.0000710925,0.00014057079,0.00083643984,0.000036118265,0.0000663064,0.00016638487,0.0038291775,0.0007233677,0.33159742,0.35961887,0.30289614],"study_design_scores_gemma":[0.000008722481,0.000026934784,0.0004164842,0.0003317607,0.0000146446355,0.00018096935,0.00004292618,0.0050028795,0.00031072326,0.29054236,0.70309615,0.000025454881],"about_ca_topic_score_codex":0.0017397528,"about_ca_topic_score_gemma":0.0029317276,"teacher_disagreement_score":0.038294394,"about_ca_system_score_codex":0.0010575746,"about_ca_system_score_gemma":0.0011194465,"threshold_uncertainty_score":0.12810755},"labels":[],"label_agreement":null},{"id":"W4212802944","doi":"10.3390/s22041393","title":"Multi-Agent Reinforcement Learning via Adaptive Kalman Temporal Difference and Successor Representation","year":2022,"lang":"en","type":"article","venue":"Sensors","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada; Ministère de la Défense Nationale","keywords":"Reinforcement learning; Overfitting; Computer science; Kalman filter; Artificial intelligence; Temporal difference learning; Representation (politics); Inefficiency; Machine learning; Artificial neural network","score_opus":0.036404439744145194,"score_gpt":0.27237387310785766,"score_spread":0.23596943336371246,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4212802944","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023883022,0.00023685672,0.97317874,0.00015710792,0.000042470117,0.000032493575,0.000025796355,0.00035789548,0.0020856725],"genre_scores_gemma":[0.9396926,0.000103714534,0.058775526,0.000064734704,0.000020624042,0.00008096314,0.00004359494,0.00002013457,0.0011980005],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99954444,0.00013393506,0.000027312366,0.000110848996,0.00012541986,0.000057975547],"domain_scores_gemma":[0.99887794,0.00060826266,0.00019607312,0.00008021341,0.00016567037,0.00007177088],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010314442,0.0006218022,0.0007878735,0.00028660914,0.00028807396,0.00059846335,0.0012192973,0.00072534574,0.0010527661],"category_scores_gemma":[0.0027449932,0.00025556184,0.00035992736,0.000263154,0.00071261614,0.00076438667,0.0009152505,0.0010242835,0.00015194534],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007034266,0.000052503954,0.00075583527,0.00005162921,0.000028620567,0.000063945554,0.000045309433,0.9460191,0.0015460392,0.0105631,0.00055467745,0.040248957],"study_design_scores_gemma":[0.0000050430995,0.000015647884,0.000037510046,0.0000016535773,0.0000020493496,0.000005519162,0.000001708249,0.9985145,0.00017629673,0.0011159302,0.00012222414,0.0000019451854],"about_ca_topic_score_codex":0.0047899247,"about_ca_topic_score_gemma":0.0037186989,"teacher_disagreement_score":0.0047899247,"about_ca_system_score_codex":0.0005932644,"about_ca_system_score_gemma":0.0009511662,"threshold_uncertainty_score":0.009524047},"labels":[],"label_agreement":null},{"id":"W4221161302","doi":"10.48550/arxiv.2201.01666","title":"Sample Efficient Deep Reinforcement Learning via Uncertainty Estimation","year":2022,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Reinforcement learning; Weighting; Computer science; Heteroscedasticity; Variance (accounting); Sample (material); Noise (video); Bayesian probability; Probabilistic logic; Process (computing); Artificial intelligence; Machine learning; Mathematical optimization; Mathematics","score_opus":0.04515281141028903,"score_gpt":0.20083283265483545,"score_spread":0.1556800212445464,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4221161302","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015221702,0.00016881239,0.9832985,0.00016162264,0.000018269053,0.00003089485,0.000023644032,0.0003615512,0.0007149266],"genre_scores_gemma":[0.8550052,0.00015652095,0.14282541,0.000171727,0.00005123916,0.00017300459,0.000116892865,0.00014968672,0.0013503865],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99859935,0.00058374816,0.00006702743,0.00025606982,0.00037429848,0.00011951963],"domain_scores_gemma":[0.9940414,0.0043053417,0.0004699261,0.0005268567,0.00048796128,0.00016846227],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030195825,0.0012632927,0.0018144369,0.0003925395,0.0004149161,0.0009819597,0.0013864798,0.0011466045,0.0015454361],"category_scores_gemma":[0.012993562,0.0007711484,0.00055435416,0.00042971736,0.0016117824,0.0021756242,0.002147041,0.0022564726,0.00028699907],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001278261,0.000060865084,0.00058753375,0.000063603584,0.00004581994,0.000035569698,0.000059016547,0.9445383,0.0010996608,0.013956727,0.0006142278,0.038810775],"study_design_scores_gemma":[0.000006794012,0.00001711432,0.000038099486,0.000003806406,0.000003307719,0.000004832677,0.0000021313347,0.9933082,0.00034132952,0.0061920183,0.00007962303,0.000002752752],"about_ca_topic_score_codex":0.0035899472,"about_ca_topic_score_gemma":0.0033014384,"teacher_disagreement_score":0.0035899472,"about_ca_system_score_codex":0.0012813105,"about_ca_system_score_gemma":0.0015305068,"threshold_uncertainty_score":0.015969276},"labels":[],"label_agreement":null},{"id":"W4224220194","doi":"10.1016/j.neunet.2022.03.037","title":"Deep learning, reinforcement learning, and world models","year":2022,"lang":"en","type":"review","venue":"Neural Networks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":493,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Artificial intelligence; Computer science; Session (web analytics); Human intelligence; Deep learning; Cognitive science; Reinforcement; Psychology","score_opus":0.04696882901226523,"score_gpt":0.2927112085831605,"score_spread":0.2457423795708953,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4224220194","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000266991,0.9915149,0.0033273532,0.0010697376,0.00027494287,0.000008070853,0.000027410051,0.000022278264,0.0034882869],"genre_scores_gemma":[0.0048736404,0.99238974,0.0010788062,0.00036991475,0.00029510082,0.000016903272,0.0000457823,0.0000039603765,0.000926211],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9998467,0.000038795904,0.00001455686,0.000030853545,0.000055234304,0.000013928244],"domain_scores_gemma":[0.9996673,0.00021395308,0.000031236526,0.000011197763,0.000056819877,0.00001951411],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000564371,0.0008358652,0.0008830496,0.0012516391,0.00020008424,0.00093713775,0.0007494772,0.001239888,0.002299385],"category_scores_gemma":[0.0010469261,0.00029897972,0.0004019927,0.0017096887,0.000669213,0.0016690531,0.0005843914,0.0017748096,0.0008937926],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000046229674,0.00007555014,0.00028895584,0.011180218,0.000110920366,0.000119484066,0.00006306879,0.005105667,0.0006332055,0.04952654,0.026339952,0.90651023],"study_design_scores_gemma":[0.000021816173,0.000114508584,0.0009933925,0.0052544703,0.00013307102,0.00074941065,0.000069920956,0.0038724935,0.00085676066,0.062099952,0.92578316,0.00005104932],"about_ca_topic_score_codex":0.0016350759,"about_ca_topic_score_gemma":0.0016968756,"teacher_disagreement_score":0.002299385,"about_ca_system_score_codex":0.00074392225,"about_ca_system_score_gemma":0.0012069109,"threshold_uncertainty_score":0.0076922774},"labels":[],"label_agreement":null},{"id":"W4224240386","doi":"10.1186/s40708-022-00156-6","title":"Hierarchical intrinsically motivated agent planning behavior with dreaming in grid environments","year":2022,"lang":"en","type":"article","venue":"Brain Informatics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Russian Foundation for Basic Research","keywords":"Grid; Psychology; Computer science; Cognitive psychology; Geography","score_opus":0.014002934279106349,"score_gpt":0.23164296966890013,"score_spread":0.2176400353897938,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4224240386","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4265214,0.00006535077,0.5664589,0.00040642923,0.000018287876,0.000036352787,0.000053027747,0.0005704215,0.005869822],"genre_scores_gemma":[0.9757488,0.000020117168,0.023250956,0.00001971757,0.000002107466,0.000017367853,0.00002055452,0.000011915937,0.0009085181],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99989927,0.000034796794,0.000005197339,0.000024900044,0.000018053528,0.00001778681],"domain_scores_gemma":[0.9997198,0.00008648057,0.00005551869,0.000059199007,0.00002741398,0.000051585175],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00027651436,0.00020241544,0.0002751461,0.000108006294,0.0003282267,0.0004978091,0.00057456706,0.00036898375,0.0012075388],"category_scores_gemma":[0.0010021057,0.0002185469,0.00031462568,0.00010135159,0.00074315415,0.00083164306,0.0008663925,0.0006118425,0.00014376936],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010110277,0.00005639288,0.0033371998,0.000039304166,0.000040449882,0.00027260702,0.00030925145,0.9275422,0.007968974,0.046765145,0.00040640973,0.013160936],"study_design_scores_gemma":[0.000007859858,0.000028103565,0.00035079574,0.0000016680333,0.0000033606752,0.000022139839,0.000025869609,0.9808044,0.0005958343,0.018004645,0.00015157925,0.000003762683],"about_ca_topic_score_codex":0.0034010587,"about_ca_topic_score_gemma":0.0044228355,"teacher_disagreement_score":0.0034010587,"about_ca_system_score_codex":0.000496602,"about_ca_system_score_gemma":0.00058979087,"threshold_uncertainty_score":0.006762564},"labels":[],"label_agreement":null},{"id":"W4225836256","doi":"10.1109/tiv.2022.3167616","title":"Deep Reinforcement Learning With NMPC Assistance Nash Switching for Urban Autonomous Driving","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Intelligent Vehicles","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Toyota Motor Corporation","keywords":"Reinforcement learning; Reinforcement; Computer science; Artificial intelligence; Psychology; Social psychology","score_opus":0.016984844672470276,"score_gpt":0.2404053362757641,"score_spread":0.22342049160329383,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4225836256","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15072586,0.0003724325,0.8408083,0.00047517463,0.00010222241,0.00009151909,0.000054160268,0.0020504487,0.005319925],"genre_scores_gemma":[0.97045904,0.000041271258,0.027840111,0.00009895173,0.0000117307045,0.00005370229,0.0000400112,0.000033852655,0.0014213179],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997471,0.00007266532,0.000011236464,0.00005244765,0.000060042305,0.000056458284],"domain_scores_gemma":[0.99916816,0.000479353,0.000081214195,0.000060942344,0.00014952358,0.000060815135],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00082123047,0.0006071951,0.00057277887,0.00024622603,0.000308135,0.00044470027,0.00091803574,0.00064289256,0.001414799],"category_scores_gemma":[0.0024497223,0.00030245102,0.0003107758,0.00019771536,0.00060181355,0.00044981236,0.0007526273,0.0011192908,0.00018856933],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009187299,0.000068402594,0.00075487845,0.000028985838,0.00001628789,0.000034807726,0.000035369292,0.9637423,0.0011408009,0.0019197236,0.0006169388,0.03154952],"study_design_scores_gemma":[0.000004619105,0.000013503221,0.00004214277,8.222474e-7,0.0000011627606,0.0000016176589,0.0000015657236,0.99935204,0.00019346224,0.00031755187,0.00007046099,0.0000011094091],"about_ca_topic_score_codex":0.012626054,"about_ca_topic_score_gemma":0.008239623,"teacher_disagreement_score":0.012626054,"about_ca_system_score_codex":0.0008030912,"about_ca_system_score_gemma":0.0012729622,"threshold_uncertainty_score":0.025105119},"labels":[],"label_agreement":null},{"id":"W4225848424","doi":"10.1609/aaai.v36i7.20764","title":"Blockwise Sequential Model Learning for Partially Observable Reinforcement Learning","year":2022,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"National Research Foundation of Korea; National Research Foundation","keywords":"Reinforcement learning; Observable; Computer science; Latent variable; Block (permutation group theory); Artificial intelligence; Artificial neural network; Markov process; Machine learning; Markov decision process; Variable (mathematics); Algorithm; Mathematics","score_opus":0.14809175079995085,"score_gpt":0.3238548945512752,"score_spread":0.17576314375132435,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4225848424","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0068768174,0.00009073036,0.99205595,0.00008720047,0.000026589823,0.000020081678,0.000026489444,0.00021378598,0.0006023544],"genre_scores_gemma":[0.7916082,0.000246586,0.20283595,0.00017814273,0.00007014209,0.00028296444,0.00023596476,0.00011659667,0.0044254856],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995104,0.00017247109,0.00002170628,0.0001112224,0.00012799632,0.00005622754],"domain_scores_gemma":[0.9991456,0.00047640022,0.00009709202,0.00008657957,0.00013729972,0.00005700054],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010962235,0.0007516913,0.001128079,0.00030992838,0.00030959497,0.0005918344,0.0013533257,0.0008977058,0.0026928214],"category_scores_gemma":[0.00276718,0.0005176463,0.00055229646,0.00033473014,0.00074297044,0.0012146606,0.0010543634,0.0016275893,0.00038491483],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005360716,0.000040010967,0.00035513312,0.000048885206,0.000032556214,0.000039079863,0.0000416469,0.9455801,0.001424288,0.021971194,0.0007156216,0.029697884],"study_design_scores_gemma":[0.000002711454,0.000008786444,0.000014368893,8.638907e-7,0.0000013072217,0.0000023261607,7.221093e-7,0.9971042,0.000102754064,0.0026524211,0.00010838584,0.00000118832],"about_ca_topic_score_codex":0.0056050858,"about_ca_topic_score_gemma":0.006285844,"teacher_disagreement_score":0.0056050858,"about_ca_system_score_codex":0.000936331,"about_ca_system_score_gemma":0.00139827,"threshold_uncertainty_score":0.0111448765},"labels":[],"label_agreement":null},{"id":"W4230186541","doi":"10.22215/etd/2007-06291","title":"Æip: a generalized framework for the study of interactive learning","year":2007,"lang":"en","type":"dissertation","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Canadian Heritage","funders":"","keywords":"Computer science; Mathematics education; Humanities; Mathematics; Art","score_opus":0.02939189543155943,"score_gpt":0.365568793608507,"score_spread":0.3361768981769476,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4230186541","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0016021361,0.00070258876,0.9911282,0.00029102713,0.00004870328,0.00003221154,0.000062952844,0.00017058587,0.0059615374],"genre_scores_gemma":[0.24190414,0.0036759563,0.7317712,0.0005364753,0.0005484531,0.0010156806,0.00039773676,0.00059157546,0.01955888],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99807274,0.00094962714,0.00008521447,0.00034838996,0.00036992255,0.00017412804],"domain_scores_gemma":[0.99765515,0.0014086341,0.0001456691,0.0004108196,0.00017248948,0.00020726591],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020307282,0.0015891024,0.0011371527,0.0012622379,0.00085780467,0.0025425027,0.003981644,0.0015384774,0.007982205],"category_scores_gemma":[0.0051066214,0.00065414555,0.0018886862,0.0019480564,0.0034858598,0.0042103827,0.0044330535,0.0038189522,0.001346136],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003092038,0.00003055659,0.00024128967,0.00017226244,0.000046882335,0.00009853663,0.0001644974,0.0382502,0.00057099154,0.9317374,0.0016803074,0.026976163],"study_design_scores_gemma":[0.000020410987,0.000052352585,0.00021200655,0.0000482618,0.000021501486,0.00007774793,0.000047676163,0.19230975,0.0002701621,0.78839254,0.018531973,0.000015681198],"about_ca_topic_score_codex":0.0042692963,"about_ca_topic_score_gemma":0.0029568581,"teacher_disagreement_score":0.007982205,"about_ca_system_score_codex":0.001543226,"about_ca_system_score_gemma":0.001630142,"threshold_uncertainty_score":0.02670312},"labels":[],"label_agreement":null},{"id":"W4231805100","doi":"10.22215/etd/2012-09679","title":"Multi-agent reinforcement learning in games","year":2012,"lang":"en","type":"dissertation","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Canadian Heritage; Library and Archives Canada","funders":"","keywords":"Reinforcement learning; Humanities; Computer science; Artificial intelligence; Art","score_opus":0.026466873854124048,"score_gpt":0.2866958448495736,"score_spread":0.26022897099544956,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4231805100","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039253194,0.0014828854,0.9426865,0.0012992971,0.00015641985,0.00012197079,0.000052864176,0.0002765788,0.014670301],"genre_scores_gemma":[0.8723326,0.0010060924,0.11023566,0.00017588993,0.0001286118,0.00033425217,0.00008434528,0.00006549875,0.015637128],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99916387,0.0005210757,0.000033657147,0.00009418644,0.00010610208,0.0000811065],"domain_scores_gemma":[0.99822384,0.0012890972,0.00010148024,0.000075364675,0.00014384156,0.00016631436],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012264806,0.0007384324,0.0008412668,0.00033201405,0.0003583181,0.0009984319,0.000936691,0.00078534294,0.0024830985],"category_scores_gemma":[0.0039031897,0.00034416482,0.0004311775,0.0003120554,0.00087267015,0.0009966898,0.0009838027,0.0014845569,0.00034245322],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011770919,0.0001408444,0.0004703045,0.00011703713,0.000065066466,0.0000649794,0.00011551529,0.85897094,0.0008009883,0.09691387,0.0025553328,0.039667513],"study_design_scores_gemma":[0.000034926285,0.000035447752,0.00006995673,0.000009351034,0.0000047895423,0.000005686865,0.000010414655,0.96735793,0.00014778718,0.031050682,0.0012688767,0.0000040871437],"about_ca_topic_score_codex":0.004888896,"about_ca_topic_score_gemma":0.004324891,"teacher_disagreement_score":0.004888896,"about_ca_system_score_codex":0.0013862256,"about_ca_system_score_gemma":0.0011434429,"threshold_uncertainty_score":0.010057807},"labels":[],"label_agreement":null},{"id":"W4233539997","doi":"10.1109/ijcnn.2006.1716136","title":"A Reinforcement Learning Framework for Medical Image Segmentation","year":2006,"lang":"en","type":"article","venue":"The 2006 IEEE International Joint Conference on Neural Network Proceedings","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Thresholding; Artificial intelligence; Segmentation; Image segmentation; Structuring element; Computer vision; Machine learning; Image (mathematics); Image processing; Mathematical morphology","score_opus":0.03411624730538505,"score_gpt":0.2948383576014699,"score_spread":0.2607221102960849,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4233539997","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0016450194,0.00023809017,0.99617755,0.00016725843,0.000032758933,0.000030942472,0.000011529698,0.0001680316,0.0015288411],"genre_scores_gemma":[0.4767704,0.0008552325,0.5120431,0.0002909112,0.00020182265,0.0004997686,0.000088163986,0.00012039985,0.009130183],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99930763,0.00028638405,0.000028432902,0.000115948336,0.0002061803,0.00005535113],"domain_scores_gemma":[0.9991522,0.0004916634,0.000082272374,0.00004600995,0.00015280647,0.0000749917],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014588152,0.0008679268,0.001136359,0.00045911747,0.00037904092,0.0009220154,0.0016437388,0.0012849573,0.0032300365],"category_scores_gemma":[0.0025390922,0.00037769298,0.0007189956,0.0004010241,0.001447723,0.0009192116,0.0010254068,0.0015337741,0.00048982416],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000057247315,0.00005018248,0.00019622198,0.000079234065,0.00003380982,0.000085906686,0.000061611056,0.9076014,0.0016425728,0.04861873,0.0010641803,0.040508855],"study_design_scores_gemma":[0.000013795104,0.000028041224,0.00003157761,0.000006489919,0.0000042576585,0.000015828777,0.0000031688246,0.9858876,0.000230498,0.01279585,0.000977444,0.0000055070823],"about_ca_topic_score_codex":0.004474459,"about_ca_topic_score_gemma":0.0028639731,"teacher_disagreement_score":0.004474459,"about_ca_system_score_codex":0.0014770805,"about_ca_system_score_gemma":0.0013711982,"threshold_uncertainty_score":0.010805547},"labels":[],"label_agreement":null},{"id":"W4238326899","doi":"10.1109/hri.2013.6483585","title":"Integrating a robot in a tabletop reservoir engineering application","year":2013,"lang":"en","type":"article","venue":"2013 8th ACM/IEEE International Conference on Human-Robot Interaction (HRI)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Visualization; Human–computer interaction; Simple (philosophy); Robot; Data visualization; Artificial intelligence","score_opus":0.11787837137254176,"score_gpt":0.3673373850222076,"score_spread":0.24945901364966583,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4238326899","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18931858,0.00031172094,0.781602,0.0004590537,0.00010759446,0.000944309,0.00029955606,0.015537589,0.011419581],"genre_scores_gemma":[0.39564207,0.0003052542,0.5900712,0.00027266843,0.00003758239,0.00039982484,0.00024677406,0.00046267733,0.012561982],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99946195,0.00010165535,0.000028215052,0.00014538367,0.00020237933,0.000060352588],"domain_scores_gemma":[0.9988939,0.0004887942,0.00008833686,0.00021170914,0.00012766861,0.00018959414],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000520961,0.00086298137,0.000585888,0.00036360082,0.000474167,0.0012829541,0.002249954,0.0011411123,0.011771584],"category_scores_gemma":[0.0018690216,0.0005659951,0.0005079134,0.00023674176,0.00064192555,0.0014714915,0.0019988385,0.00071100093,0.0024111257],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017025121,0.0012690491,0.007114353,0.0015987392,0.00016795653,0.004328245,0.0036841608,0.031238731,0.51478124,0.0056409254,0.011284509,0.4171897],"study_design_scores_gemma":[0.00051582523,0.006166015,0.017802032,0.00039808522,0.000299576,0.007233339,0.0019055835,0.51462275,0.2275442,0.004411416,0.21855925,0.00054202386],"about_ca_topic_score_codex":0.0013196548,"about_ca_topic_score_gemma":0.0018119533,"teacher_disagreement_score":0.011771584,"about_ca_system_score_codex":0.00019273035,"about_ca_system_score_gemma":0.00070466445,"threshold_uncertainty_score":0.039379895},"labels":[],"label_agreement":null},{"id":"W4240203305","doi":"10.22215/etd/2014-10293","title":"Multiple Model Reinforcement Learning for Environments with Poissonian Time Delays","year":2014,"lang":"en","type":"dissertation","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Reinforcement learning; Computer science; Convergence (economics); Mobile robot; Robot; Reinforcement; Q-learning; Artificial intelligence; Grid; Simulation; Mathematics; Engineering","score_opus":0.01021222507287907,"score_gpt":0.22475181885179654,"score_spread":0.21453959377891746,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4240203305","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018495284,0.00022704694,0.9792046,0.00022743508,0.000047806334,0.000029864615,0.000018166535,0.00018781432,0.0015619315],"genre_scores_gemma":[0.8692905,0.00034208148,0.12297507,0.00014355237,0.000081730825,0.00020592463,0.00008603459,0.000090660906,0.0067844857],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99924266,0.00028384037,0.000031486245,0.00015829284,0.0001673911,0.00011639502],"domain_scores_gemma":[0.99743086,0.0018508409,0.0002400588,0.00009598108,0.00023471106,0.00014756626],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017064909,0.0007917541,0.0013347497,0.00037724877,0.00045362485,0.00093901256,0.0016734974,0.00092412526,0.0019406158],"category_scores_gemma":[0.006065129,0.0005476943,0.0007172983,0.00038510183,0.0009335086,0.0012018466,0.0016300495,0.0018580209,0.0002674734],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000037330272,0.000027944605,0.00023169245,0.000028806131,0.000019385741,0.000036631707,0.000033415152,0.9725002,0.0002686332,0.015200902,0.00031491095,0.011300127],"study_design_scores_gemma":[0.000007905532,0.000008752262,0.000020120176,0.0000016608218,0.0000018961881,0.0000035617534,0.000002329719,0.9940526,0.0000615921,0.005720955,0.00011674492,0.0000019101635],"about_ca_topic_score_codex":0.0064011426,"about_ca_topic_score_gemma":0.0048827496,"teacher_disagreement_score":0.0064011426,"about_ca_system_score_codex":0.0015888679,"about_ca_system_score_gemma":0.0014325695,"threshold_uncertainty_score":0.012727797},"labels":[],"label_agreement":null},{"id":"W4243101510","doi":"10.1002/9781118445112.stat08000","title":"Reinforcement Learning","year":2017,"lang":"en","type":"other","venue":"Wiley StatsRef: Statistics Reference Online","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Reinforcement learning; Computer science; A priori and a posteriori; Sequence (biology); Artificial intelligence; Quality (philosophy); Reinforcement; Machine learning; Psychology","score_opus":0.04106047788657132,"score_gpt":0.31309103557408263,"score_spread":0.27203055768751133,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4243101510","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0072893496,0.0044834265,0.9266406,0.003061377,0.00052702473,0.00015500883,0.00027986974,0.0010827616,0.056480553],"genre_scores_gemma":[0.70912844,0.007607078,0.21810836,0.002074705,0.0010255447,0.00064124883,0.0012211431,0.0004201432,0.059773322],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.998672,0.0005106701,0.00006816159,0.00031968043,0.00033795493,0.00009144806],"domain_scores_gemma":[0.9975503,0.0014798031,0.00017004734,0.0002480981,0.00040363733,0.000148175],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013366045,0.0010357191,0.0010076746,0.00049086346,0.00035487505,0.00172318,0.0014652789,0.0010293502,0.015792511],"category_scores_gemma":[0.006806139,0.00023607751,0.0005190938,0.0005338096,0.001331566,0.0014366502,0.0013631241,0.0019756812,0.003351572],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015610809,0.00018380333,0.0013428354,0.00051881006,0.00013746829,0.00015605442,0.00010196213,0.32721263,0.0015786069,0.35841882,0.032417454,0.2777754],"study_design_scores_gemma":[0.00007729536,0.00011601216,0.00034781272,0.00016184546,0.000038081445,0.00010765563,0.000034550758,0.6579032,0.0010477763,0.29476428,0.045367606,0.000033869565],"about_ca_topic_score_codex":0.0016576224,"about_ca_topic_score_gemma":0.0012890078,"teacher_disagreement_score":0.015792511,"about_ca_system_score_codex":0.0010823512,"about_ca_system_score_gemma":0.0013838955,"threshold_uncertainty_score":0.052831233},"labels":[],"label_agreement":null},{"id":"W4244090038","doi":"10.1007/978-1-4471-7452-3_17","title":"Reinforcement Learning","year":2019,"lang":"en","type":"book-chapter","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Reinforcement learning; Reinforcement; Computer science; Error-driven learning; Formalism (music); Artificial intelligence; Psychology; Social psychology","score_opus":0.01983590515921726,"score_gpt":0.22694637773350715,"score_spread":0.20711047257428988,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4244090038","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018417051,0.008910156,0.24822135,0.0017173439,0.0011210804,0.0000836572,0.00027227864,0.0015385997,0.73629373],"genre_scores_gemma":[0.053726237,0.008980601,0.05871386,0.00077021465,0.000507107,0.00016515088,0.00055936736,0.00036793872,0.87620956],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9998542,0.000020046955,0.0000043223877,0.000034104076,0.00007666907,0.000010701093],"domain_scores_gemma":[0.9998684,0.00004849445,0.0000067649785,0.000026954649,0.000035810703,0.000013555872],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00018153629,0.0007733561,0.00037858612,0.00041876273,0.00030755388,0.0010244391,0.0006618498,0.00066486,0.055845726],"category_scores_gemma":[0.00070279837,0.00022621188,0.00024606928,0.0004601758,0.00057453173,0.0010428494,0.0007633425,0.0013379991,0.0223146],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002618182,0.00007401361,0.000090614376,0.00019556287,0.000012646824,0.000035053497,0.00005613989,0.011150932,0.0017030453,0.18068637,0.14343306,0.6625364],"study_design_scores_gemma":[0.000012739326,0.000054694345,0.000297166,0.00020065231,0.000011030942,0.00018992343,0.00003871274,0.031600628,0.0026930617,0.22647865,0.73840094,0.000021762278],"about_ca_topic_score_codex":0.0009202816,"about_ca_topic_score_gemma":0.0015664868,"teacher_disagreement_score":0.055845726,"about_ca_system_score_codex":0.00076109625,"about_ca_system_score_gemma":0.00057090825,"threshold_uncertainty_score":0.18682253},"labels":[],"label_agreement":null},{"id":"W4246058920","doi":"10.1109/.2005.1507451","title":"Attention shifts during action sequence recognition for social robots","year":2005,"lang":"en","type":"article","venue":"ICAR '05. Proceedings., 12th International Conference on Advanced Robotics, 2005.","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Imperial College of Toronto","funders":"Engineering and Physical Sciences Research Council; Royal Society","keywords":"Action (physics); Sequence (biology); Robot; Computer science; Artificial intelligence; Human–computer interaction; Cognitive science; Psychology; Physics; Biology; Genetics","score_opus":0.10749375011927112,"score_gpt":0.34372097079146,"score_spread":0.2362272206721889,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4246058920","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7225999,0.00033801762,0.27214405,0.00030983385,0.00007557139,0.000118071985,0.000030058014,0.001985187,0.0023994003],"genre_scores_gemma":[0.9829222,0.000026990661,0.016463302,0.000040674502,0.000008475334,0.000022134302,0.000014458747,0.000022973181,0.00047883607],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99970645,0.0000829725,0.0000109805305,0.000075372314,0.000074564996,0.000049626105],"domain_scores_gemma":[0.99797004,0.0012859991,0.00023052363,0.00013775991,0.00020381738,0.00017180225],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008603733,0.00033574007,0.00047251896,0.00036309336,0.0003339416,0.00035979922,0.0008010791,0.00065617147,0.0017132129],"category_scores_gemma":[0.005850565,0.00033951615,0.00022503259,0.00017398054,0.00053859793,0.0007680671,0.0008178428,0.0006834034,0.00022400549],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016896634,0.00048234826,0.010316135,0.00020592696,0.0001238154,0.00056564715,0.001601961,0.08911043,0.3752934,0.0046170694,0.001810473,0.5141831],"study_design_scores_gemma":[0.000059849524,0.00043199502,0.013679066,0.00001051181,0.000033600343,0.0001815226,0.00016221612,0.9309935,0.04728811,0.0062866663,0.0008455895,0.000027434055],"about_ca_topic_score_codex":0.0044540786,"about_ca_topic_score_gemma":0.004030083,"teacher_disagreement_score":0.0044540786,"about_ca_system_score_codex":0.0008229582,"about_ca_system_score_gemma":0.00047656702,"threshold_uncertainty_score":0.008856297},"labels":[],"label_agreement":null},{"id":"W4247188402","doi":"10.36227/techrxiv.14842245","title":"Safe Deployment of a Reinforcement Learning Robot Using Self Stabilization","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Software deployment; Robot; Computer science; Robotics; Artificial intelligence; State space; Simulation; Human–computer interaction; Software engineering; Mathematics","score_opus":0.03484376922820494,"score_gpt":0.2772683896775365,"score_spread":0.24242462044933155,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4247188402","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09211677,0.00006531234,0.9028169,0.00017861815,0.00003816079,0.00010407751,0.000015395783,0.0017610763,0.0029036724],"genre_scores_gemma":[0.95708275,0.000022901037,0.04127671,0.000042259824,0.000007087296,0.0000801342,0.000020028087,0.000058327292,0.0014099061],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995055,0.0001290829,0.000025634045,0.00011551333,0.0001489581,0.0000753494],"domain_scores_gemma":[0.99844044,0.0006095673,0.00029645333,0.00025463966,0.00024650857,0.00015232238],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00077126164,0.0007352357,0.00038773663,0.00026472806,0.0003303769,0.00044752506,0.00088469236,0.00078284665,0.0014932189],"category_scores_gemma":[0.0029837636,0.0002790957,0.0002899269,0.00010155069,0.0011635432,0.00051016186,0.0011932976,0.00084773457,0.00041382114],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018023935,0.00009347493,0.00077170605,0.00006179553,0.000021124539,0.00022952666,0.00016980918,0.93703353,0.023332097,0.005580891,0.00060655986,0.03191924],"study_design_scores_gemma":[0.000015765874,0.000091962276,0.000080353326,0.000004387739,0.0000029629302,0.000016935877,0.000008530687,0.9958527,0.002776856,0.0009033252,0.00024233338,0.0000038546505],"about_ca_topic_score_codex":0.003219161,"about_ca_topic_score_gemma":0.0021820944,"teacher_disagreement_score":0.003219161,"about_ca_system_score_codex":0.00052728236,"about_ca_system_score_gemma":0.00096245314,"threshold_uncertainty_score":0.0064008236},"labels":[],"label_agreement":null},{"id":"W4250455569","doi":"10.1109/ijcnn.2006.1716795","title":"Extend Single-agent Reinforcement Learning Approach to a Multi-robot Cooperative Task in an Unknown Dynamic Environment","year":2006,"lang":"en","type":"article","venue":"The 2006 IEEE International Joint Conference on Neural Network Proceedings","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Reinforcement learning; Computer science; Robot; Markov decision process; Robustness (evolution); Robot learning; Artificial intelligence; Q-learning; Obstacle; Mobile robot; Markov process; Machine learning; Mathematics","score_opus":0.052808010837956824,"score_gpt":0.26725047299880855,"score_spread":0.21444246216085172,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4250455569","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016313445,0.00011915279,0.9818392,0.00017255326,0.000033178003,0.00003625384,0.000009320058,0.00015281467,0.0013240209],"genre_scores_gemma":[0.8544557,0.00023207067,0.14244008,0.0001464443,0.000041930358,0.00014223858,0.000030434083,0.000028778053,0.002482333],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996356,0.00013504444,0.00002068958,0.00008474714,0.00008204321,0.000041871135],"domain_scores_gemma":[0.99926907,0.00041149085,0.00007389131,0.000075684424,0.00011631343,0.000053640255],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009824346,0.0006061138,0.0007484176,0.00021911402,0.0003270694,0.00040712536,0.0009693963,0.0007411573,0.0013827368],"category_scores_gemma":[0.001574774,0.0002174507,0.00057565817,0.00022235679,0.0007291179,0.0008762276,0.00076441653,0.0009221503,0.00025111233],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000046619578,0.00008590785,0.00074913265,0.0000818676,0.000050052106,0.00022339486,0.00010132473,0.9375379,0.0040827133,0.014020315,0.00040040957,0.042620383],"study_design_scores_gemma":[0.000009644513,0.000035189852,0.000057993926,0.0000022162692,0.0000048472766,0.000019644414,0.000004276373,0.99501,0.00045176139,0.004001485,0.00039938354,0.0000034290551],"about_ca_topic_score_codex":0.0021157216,"about_ca_topic_score_gemma":0.0012105989,"teacher_disagreement_score":0.0021157216,"about_ca_system_score_codex":0.0004123751,"about_ca_system_score_gemma":0.00083136064,"threshold_uncertainty_score":0.0051956773},"labels":[],"label_agreement":null},{"id":"W4251839393","doi":"10.1007/978-1-4899-7502-7_77-1","title":"Dynamic Programming","year":2014,"lang":"en","type":"book-chapter","venue":"Encyclopedia of Machine Learning and Data Mining","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa; University of British Columbia","funders":"","keywords":"Computer science; Programming language","score_opus":0.016937534042023732,"score_gpt":0.26894804897153735,"score_spread":0.2520105149295136,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4251839393","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0020033512,0.005746179,0.66921127,0.0016916808,0.0006127212,0.0000677681,0.0008044293,0.0010424509,0.3188201],"genre_scores_gemma":[0.17860746,0.013703691,0.33628127,0.0015734724,0.0010680985,0.00067525735,0.003451149,0.0015528415,0.46308675],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9997292,0.000053783206,0.000011788379,0.00008775703,0.00009218872,0.000025188167],"domain_scores_gemma":[0.99973863,0.00011928411,0.000014923152,0.0000405286,0.00006454125,0.000022034492],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003205506,0.0010436296,0.0008091932,0.0004804275,0.0004059991,0.0020214145,0.0008745077,0.00072344835,0.043830648],"category_scores_gemma":[0.001200647,0.0003782293,0.000514314,0.00076651806,0.0006992275,0.0011049912,0.0010966068,0.0019265714,0.015697107],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000368505,0.0000724247,0.00018497823,0.00022214286,0.00003348563,0.00005330854,0.0000616892,0.03129702,0.000980109,0.6248428,0.077567175,0.26464796],"study_design_scores_gemma":[0.000027295531,0.000029670446,0.00019771217,0.00011843304,0.00001831026,0.00011998621,0.00003289642,0.092743345,0.00087656046,0.6953681,0.21044646,0.000021266746],"about_ca_topic_score_codex":0.0011818334,"about_ca_topic_score_gemma":0.0014867903,"teacher_disagreement_score":0.043830648,"about_ca_system_score_codex":0.00073465303,"about_ca_system_score_gemma":0.0007690847,"threshold_uncertainty_score":0.1466282},"labels":[],"label_agreement":null},{"id":"W4254092108","doi":"10.1145/860685.860689","title":"Coordination in multiagent reinforcement learning","year":2003,"lang":"en","type":"article","venue":"Proceedings of the second international joint conference on Autonomous agents and multiagent systems - AAMAS '03","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Multi-agent system; Artificial intelligence","score_opus":0.03543263955971353,"score_gpt":0.256654267176619,"score_spread":0.22122162761690545,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4254092108","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02632375,0.0007294775,0.96328473,0.0009554277,0.00007565797,0.00007112318,0.000047540827,0.00023488149,0.008277339],"genre_scores_gemma":[0.9259737,0.0004087777,0.06975681,0.00020882941,0.00006680155,0.00024759243,0.000057138182,0.000047333942,0.0032329871],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99798715,0.001098291,0.00009246839,0.00033126332,0.00031224548,0.00017854168],"domain_scores_gemma":[0.99551505,0.0029240951,0.00058588164,0.0002952205,0.0003754063,0.0003043859],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028616362,0.0008461661,0.0012360482,0.00048376122,0.00060689275,0.0012687112,0.0014359881,0.0014005323,0.0021318214],"category_scores_gemma":[0.010686688,0.00045733727,0.00048692164,0.00045194672,0.0020455283,0.0017023919,0.0015487265,0.0016769998,0.00026872254],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005756796,0.00004223245,0.00067117205,0.00006615865,0.00004652069,0.00010338375,0.00011743247,0.85469323,0.00042934902,0.13053603,0.0008833896,0.012353524],"study_design_scores_gemma":[0.00002656941,0.000027922046,0.000091321344,0.000009085716,0.000005695236,0.000014545972,0.000015179331,0.9145584,0.00010721452,0.0843997,0.00073807547,0.000006293551],"about_ca_topic_score_codex":0.004682449,"about_ca_topic_score_gemma":0.002945678,"teacher_disagreement_score":0.004682449,"about_ca_system_score_codex":0.0020277414,"about_ca_system_score_gemma":0.0015097369,"threshold_uncertainty_score":0.015133977},"labels":[],"label_agreement":null},{"id":"W4255384105","doi":"10.31234/osf.io/ptz2r","title":"Inferring Actions, Intentions, and Causal Relations in a Neural Network","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Institute for Advanced Research","funders":"Wellcome Trust; Canadian Institute for Advanced Research","keywords":"Generative grammar; Inference; Action (physics); Computer science; Representation (politics); Artificial intelligence; Process (computing); Generative model; Causal inference; Artificial neural network; Control (management); Machine learning; Cognitive science; Theoretical computer science; Psychology; Mathematics; Econometrics","score_opus":0.04283485095793588,"score_gpt":0.2884694462248224,"score_spread":0.24563459526688652,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4255384105","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16389683,0.00032934692,0.8301917,0.00097632484,0.00004028135,0.00003739102,0.00017420397,0.00043543792,0.0039184927],"genre_scores_gemma":[0.92165977,0.00024717354,0.07475568,0.000112243484,0.000021043938,0.000058681882,0.0001358798,0.000026613729,0.002982867],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997019,0.00010022343,0.000012237666,0.000102611404,0.000043672404,0.00003936888],"domain_scores_gemma":[0.99927,0.0005211034,0.000071244125,0.000044624696,0.000052296138,0.000040731687],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00063730264,0.00041670716,0.00040447857,0.0004015463,0.0003181618,0.0008496677,0.000845757,0.0011746689,0.0014972079],"category_scores_gemma":[0.002945734,0.0005630017,0.0004142867,0.00036382253,0.0010039136,0.0013846568,0.00076410046,0.0013186975,0.00018613682],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000096838616,0.000047048605,0.0018464272,0.00004031983,0.000042993113,0.0001139385,0.00010568284,0.9336429,0.0030640834,0.029407997,0.00049534114,0.03109649],"study_design_scores_gemma":[0.0000040201408,0.0000062077324,0.00014747147,0.0000032260994,0.000004637925,0.0000069532857,0.000005114383,0.9850701,0.00027329125,0.014364143,0.00011139185,0.0000034935017],"about_ca_topic_score_codex":0.011302164,"about_ca_topic_score_gemma":0.012719432,"teacher_disagreement_score":0.011302164,"about_ca_system_score_codex":0.00096184044,"about_ca_system_score_gemma":0.0007063958,"threshold_uncertainty_score":0.022472799},"labels":[],"label_agreement":null},{"id":"W4256006693","doi":"10.1109/wsc.2015.7408303","title":"Imitation challenges: From uniform random variables to complex systems","year":2015,"lang":"en","type":"article","venue":"2015 Winter Simulation Conference (WSC)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Imitation; Computer science; Random variable; Theoretical computer science; Mathematics; Statistics; Psychology","score_opus":0.1790954619224176,"score_gpt":0.3286150655444003,"score_spread":0.1495196036219827,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4256006693","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018135382,0.00447496,0.94333017,0.013358529,0.00059014774,0.000065292166,0.00013525819,0.0002996536,0.019610673],"genre_scores_gemma":[0.8166654,0.0065453523,0.15904419,0.002712524,0.0013564832,0.0004357033,0.00020345055,0.00034012698,0.012696788],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99734443,0.0015523668,0.00008411875,0.00038154528,0.00050635025,0.00013113832],"domain_scores_gemma":[0.98796785,0.009802971,0.00049336045,0.0009540494,0.0004005952,0.0003812318],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035949294,0.0009307279,0.0013875135,0.00065062236,0.0010661971,0.0032659401,0.00212234,0.0027492258,0.00510217],"category_scores_gemma":[0.027940571,0.0006881537,0.001087246,0.00061679573,0.0060989377,0.0071276547,0.003669058,0.00484546,0.0005837891],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000025650188,0.000014703614,0.0004919348,0.00009235235,0.000034663157,0.0001455898,0.00021900561,0.07359363,0.00017613832,0.91359246,0.0025539312,0.0090599],"study_design_scores_gemma":[0.000013120621,0.000017676735,0.000113542636,0.000035718713,0.0000060674834,0.000059165755,0.0000507582,0.18390928,0.00014527366,0.81028974,0.0053452626,0.000014384452],"about_ca_topic_score_codex":0.0019937474,"about_ca_topic_score_gemma":0.00088961323,"teacher_disagreement_score":0.00510217,"about_ca_system_score_codex":0.0018351191,"about_ca_system_score_gemma":0.001035841,"threshold_uncertainty_score":0.019012034},"labels":[],"label_agreement":null},{"id":"W4280563754","doi":"10.18196/jrc.v3i2.13082","title":"Consensus of Multi-agent Reinforcement Learning Systems: The Effect of Immediate Rewards","year":2022,"lang":"en","type":"article","venue":"Journal of Robotics and Control (JRC)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Reinforcement learning; Cumulative prospect theory; Computer science; Function (biology); Artificial intelligence; Mathematics; Statistics","score_opus":0.010502809948478941,"score_gpt":0.23320032861499804,"score_spread":0.2226975186665191,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4280563754","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21421143,0.00048119333,0.78114,0.000378468,0.00006685637,0.000062249994,0.0000145215945,0.00022039902,0.0034248582],"genre_scores_gemma":[0.9890843,0.00007044284,0.010252726,0.000027894674,0.000012813519,0.000019948395,0.0000045600946,0.00001134083,0.00051598705],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989293,0.0004836775,0.00004233703,0.00017822483,0.00019724021,0.00016915104],"domain_scores_gemma":[0.99219644,0.0056605632,0.0008337416,0.0002768709,0.00064624276,0.00038614115],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028792967,0.0007890377,0.0009199568,0.0003404407,0.0004353296,0.00069733994,0.00076850347,0.00080288725,0.00079750473],"category_scores_gemma":[0.008835212,0.0002790345,0.00043173152,0.0001739613,0.0011217395,0.001106039,0.001020599,0.0009905127,0.0000825613],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009545744,0.000036567093,0.0005771861,0.00004458989,0.000032217304,0.000087437584,0.000046836172,0.9858242,0.0013968077,0.0052270014,0.000098003846,0.006533733],"study_design_scores_gemma":[0.000010262239,0.00006553176,0.00011249371,0.0000032162452,0.0000058216024,0.000010257308,0.000009404567,0.99757785,0.0004552673,0.0016886607,0.000057591285,0.0000037363648],"about_ca_topic_score_codex":0.0024741527,"about_ca_topic_score_gemma":0.0013493121,"teacher_disagreement_score":0.0028792967,"about_ca_system_score_codex":0.0008335281,"about_ca_system_score_gemma":0.00089590205,"threshold_uncertainty_score":0.015227318},"labels":[],"label_agreement":null},{"id":"W4283160467","doi":"10.1002/cjce.24508","title":"A survey and comparative evaluation of actor‐critic methods in process control","year":2022,"lang":"en","type":"article","venue":"The Canadian Journal of Chemical Engineering","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Robustness (evolution); Computer science; Process (computing); Optimal control; Artificial neural network; Control engineering; Process control; Control (management); Artificial intelligence; Machine learning; Mathematical optimization; Engineering; Mathematics","score_opus":0.0722343985281289,"score_gpt":0.3391115488946168,"score_spread":0.2668771503664879,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4283160467","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020690197,0.07607859,0.88018775,0.0009951253,0.0003670237,0.00017721062,0.00009079017,0.0008501045,0.020563282],"genre_scores_gemma":[0.73479605,0.050400194,0.2095347,0.00029321172,0.00044460516,0.00026730503,0.00021175494,0.00026824785,0.0037839133],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99598026,0.0019861253,0.00030854862,0.00045895766,0.0011608751,0.00010525687],"domain_scores_gemma":[0.9895752,0.007965489,0.00039306082,0.00047858132,0.0014351177,0.00015256953],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006411848,0.0015219054,0.0016572208,0.0017653308,0.0003705501,0.0019166187,0.0020522967,0.0016110926,0.001888859],"category_scores_gemma":[0.010156903,0.000733524,0.0009327608,0.0020596872,0.0010631902,0.0012235623,0.000956565,0.0016198934,0.0003903474],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026127396,0.00016228874,0.0011774143,0.0018156646,0.00032873204,0.00004359557,0.00009846492,0.7003798,0.0010609961,0.021998735,0.0012786832,0.27139434],"study_design_scores_gemma":[0.00002974213,0.00018620415,0.0004523529,0.00020346178,0.00004814778,0.000027001348,0.00002983736,0.98833746,0.0012021676,0.004386341,0.0050762882,0.00002103168],"about_ca_topic_score_codex":0.006931486,"about_ca_topic_score_gemma":0.0025572316,"teacher_disagreement_score":0.006931486,"about_ca_system_score_codex":0.001453714,"about_ca_system_score_gemma":0.0011402463,"threshold_uncertainty_score":0.0339095},"labels":[],"label_agreement":null},{"id":"W4283797949","doi":"10.1609/aaai.v36i9.21233","title":"Stochastic Goal Recognition Design Problems with Suboptimal Agents","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"National Science Foundation","keywords":"Ambiguity; Computer science; Generalization; Observability; Action (physics); Benchmark (surveying); Artificial intelligence; Range (aeronautics); Mathematical optimization; Observer (physics); Machine learning; Mathematics; Engineering","score_opus":0.12589060475098662,"score_gpt":0.27620741389932374,"score_spread":0.15031680914833712,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4283797949","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02878604,0.00019111793,0.9682109,0.00032701803,0.000027181539,0.00009602345,0.00005120484,0.00014664074,0.0021637888],"genre_scores_gemma":[0.77943194,0.00020970411,0.21597113,0.00026699767,0.000039031303,0.00039465225,0.0001809194,0.00007600863,0.0034297307],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9967443,0.0012747523,0.0002544877,0.00085495133,0.0005395367,0.00033191298],"domain_scores_gemma":[0.9903702,0.0067091817,0.0011940624,0.0005818868,0.0007517188,0.0003929081],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004409973,0.0015235005,0.0017380168,0.0006875556,0.00064747635,0.0015808986,0.0013541837,0.002257133,0.0023528433],"category_scores_gemma":[0.01344803,0.0010002409,0.0014447276,0.00048100392,0.0021091078,0.0016916889,0.002409604,0.0020672292,0.00030051605],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007635644,0.000037863614,0.0006466106,0.000085054475,0.00004034606,0.00008578094,0.00007757426,0.9611997,0.00091422733,0.025116716,0.00031785766,0.011402037],"study_design_scores_gemma":[0.000027076147,0.00005664916,0.00009688965,0.000012365358,0.000011707954,0.000019560275,0.000017776574,0.9761681,0.00047481546,0.022749515,0.0003566661,0.000008837965],"about_ca_topic_score_codex":0.0035670502,"about_ca_topic_score_gemma":0.0026752914,"teacher_disagreement_score":0.004409973,"about_ca_system_score_codex":0.002000957,"about_ca_system_score_gemma":0.0018061317,"threshold_uncertainty_score":0.023322403},"labels":[],"label_agreement":null},{"id":"W4283801867","doi":"10.1609/aaai.v36i7.20764","title":"Blockwise Sequential Model Learning for Partially Observable Reinforcement Learning","year":2022,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"National Research Foundation of Korea; National Research Foundation","keywords":"Observable; Reinforcement learning; Computer science; Latent variable; Block (permutation group theory); Artificial intelligence; Artificial neural network; Markov process; Variable (mathematics); Machine learning; Markov chain; Algorithm; Mathematics","score_opus":0.0468047699607824,"score_gpt":0.2675118803325101,"score_spread":0.22070711037172772,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4283801867","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007072716,0.00009552078,0.9918677,0.00008096027,0.000026968375,0.000019785619,0.000024697354,0.00022050666,0.0005910659],"genre_scores_gemma":[0.8208391,0.00023520763,0.17422001,0.000165328,0.00006279052,0.00025314937,0.00021013555,0.00010061944,0.003913592],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995372,0.00015353099,0.00002127224,0.00010971069,0.00012186582,0.000056272824],"domain_scores_gemma":[0.9992274,0.00041466576,0.00009442601,0.00007646989,0.00013347057,0.000053535092],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010027818,0.0007715719,0.00113126,0.0003088909,0.00030354183,0.00055433245,0.0013394278,0.00084521645,0.002510194],"category_scores_gemma":[0.0024255891,0.0004980113,0.000543679,0.0003238453,0.00072806806,0.0011335269,0.0010026423,0.0015391903,0.0003617559],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000053934782,0.000040468934,0.00038274954,0.000052105515,0.000033778833,0.000041523796,0.000041311137,0.9466303,0.0016324099,0.018430399,0.000714279,0.031946804],"study_design_scores_gemma":[0.0000026824544,0.000009788532,0.000015751832,8.8928573e-7,0.0000014228686,0.0000025161598,7.241877e-7,0.99764615,0.00011026664,0.0021039161,0.000104573446,0.000001245707],"about_ca_topic_score_codex":0.0057193385,"about_ca_topic_score_gemma":0.006396289,"teacher_disagreement_score":0.0057193385,"about_ca_system_score_codex":0.00085412705,"about_ca_system_score_gemma":0.0014034858,"threshold_uncertainty_score":0.011372089},"labels":[],"label_agreement":null},{"id":"W4283815468","doi":"10.1609/aaai.v36i6.20639","title":"A Generalized Bootstrap Target for Value-Learning, Efficiently Combining Value and Feature Predictions","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Institute for Advanced Research; McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"Fonds de recherche du Québec – Nature et technologies; Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Bootstrapping (finance); Computer science; Reinforcement learning; Successor cardinal; Value (mathematics); Artificial intelligence; Bellman equation; Machine learning; Function (biology); Generality; Feature (linguistics); Mathematics; Econometrics; Mathematical optimization","score_opus":0.05540860206408431,"score_gpt":0.2917252413087566,"score_spread":0.23631663924467228,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4283815468","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0055186446,0.000054861754,0.9932458,0.00008843128,0.000015382917,0.000030674444,0.000025139132,0.00043805497,0.00058303896],"genre_scores_gemma":[0.48161668,0.00012843766,0.5148178,0.00021266722,0.00008660222,0.00034947874,0.00025967165,0.00036094067,0.002167726],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977884,0.00083404797,0.000103627775,0.00045619925,0.0006664688,0.0001512652],"domain_scores_gemma":[0.99230266,0.0045664525,0.0005706104,0.0013409415,0.000944791,0.00027452176],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051881354,0.0013277441,0.0017483075,0.0008848289,0.0005041694,0.0018974049,0.0034608392,0.0018339396,0.003488737],"category_scores_gemma":[0.026638178,0.000851788,0.0009731524,0.0009636231,0.0017514344,0.0045461417,0.0037996122,0.0032085546,0.0010964036],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036398118,0.00024316223,0.0028919692,0.00016571481,0.000150352,0.00013495293,0.00022558394,0.682734,0.0063243755,0.08194112,0.0029700492,0.22185467],"study_design_scores_gemma":[0.000007857966,0.000040732488,0.00010178779,0.000010857704,0.000008485288,0.000021217784,0.0000055816313,0.981124,0.0013513572,0.016805688,0.00051325624,0.000009210896],"about_ca_topic_score_codex":0.0021459074,"about_ca_topic_score_gemma":0.002051418,"teacher_disagreement_score":0.0051881354,"about_ca_system_score_codex":0.0013403038,"about_ca_system_score_gemma":0.0016236727,"threshold_uncertainty_score":0.027437806},"labels":[],"label_agreement":null},{"id":"W4285061979","doi":"","title":"Graph augmented Deep Reinforcement Learning in the GameRLand3D environment","year":2021,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ubisoft (Canada)","funders":"","keywords":"Reinforcement learning; Graph; Computer science; Artificial intelligence; Reinforcement; Theoretical computer science; Psychology; Social psychology","score_opus":0.012907675169355612,"score_gpt":0.2150617302733661,"score_spread":0.20215405510401047,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285061979","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15334916,0.00077862793,0.8215376,0.00096248864,0.00019596856,0.00016588534,0.00097069645,0.010554612,0.011484934],"genre_scores_gemma":[0.7759751,0.00019648526,0.21766837,0.00028351898,0.000021758427,0.00017311609,0.00090880727,0.0004451859,0.004327682],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99970716,0.00009995134,0.00000780069,0.00008339397,0.000054477463,0.00004727301],"domain_scores_gemma":[0.999569,0.00024705852,0.000028905708,0.000060627837,0.000045432538,0.000048935333],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004833837,0.0012013351,0.00067186216,0.00030704617,0.00039346004,0.0008228586,0.0014150665,0.0012767713,0.0027812805],"category_scores_gemma":[0.0018708194,0.00042630214,0.0006039686,0.00028973512,0.0010748253,0.00096736377,0.0013914913,0.0015445859,0.000531634],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000950826,0.000053586275,0.00031008257,0.000049914,0.000021246877,0.00009783662,0.00003509508,0.974686,0.0012714755,0.0034994252,0.0016765831,0.018203652],"study_design_scores_gemma":[0.000016096761,0.000023801043,0.000060821054,0.000003744567,0.0000024558033,0.000008612114,0.0000065165927,0.99544466,0.00065356254,0.003096548,0.0006788594,0.0000042711144],"about_ca_topic_score_codex":0.020258311,"about_ca_topic_score_gemma":0.02536617,"teacher_disagreement_score":0.020258311,"about_ca_system_score_codex":0.0011960799,"about_ca_system_score_gemma":0.0012695997,"threshold_uncertainty_score":0.04028076},"labels":[],"label_agreement":null},{"id":"W4285102525","doi":"10.1109/icra46639.2022.9811652","title":"Exploiting Abstract Symmetries in Reinforcement Learning for Complex Environments","year":2022,"lang":"en","type":"article","venue":"2022 International Conference on Robotics and Automation (ICRA)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia, Okanagan Campus; University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Computer science; Heuristics; Abstraction; Exploit; Sample (material); Artificial intelligence; Sample complexity; State space; Inefficiency; Space (punctuation); Field (mathematics); State (computer science); Algorithm; Mathematics","score_opus":0.062108136455817274,"score_gpt":0.29000277259952806,"score_spread":0.22789463614371078,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285102525","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015453129,0.00008919811,0.98334897,0.00011987977,0.000014433085,0.00003529188,0.0000113914075,0.00016086006,0.0007667971],"genre_scores_gemma":[0.68573725,0.00025129886,0.3125244,0.000109734945,0.00003888388,0.00017251105,0.000051369625,0.000062100604,0.0010523323],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99928516,0.00032340604,0.0000509768,0.00010502161,0.0001750144,0.00006048774],"domain_scores_gemma":[0.9976751,0.0012476735,0.00031182857,0.0004579904,0.0001275744,0.00017972387],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019633668,0.0006032199,0.0008847088,0.000380572,0.00034950473,0.00092727493,0.00090269995,0.0005960043,0.001090993],"category_scores_gemma":[0.0058333203,0.00034765704,0.0007340816,0.00030638356,0.002026792,0.0023404185,0.002701586,0.0019598932,0.00019379344],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016570473,0.00009774658,0.0016565174,0.00015325531,0.000062202176,0.00016457819,0.00030274445,0.6756751,0.0066488637,0.21353585,0.0006581088,0.10087944],"study_design_scores_gemma":[0.00003748303,0.00010012247,0.00019185536,0.0000143935595,0.000009784394,0.000047529535,0.000020776199,0.85453457,0.0016829958,0.14228265,0.0010613474,0.000016434371],"about_ca_topic_score_codex":0.0009296409,"about_ca_topic_score_gemma":0.0009558754,"teacher_disagreement_score":0.0019633668,"about_ca_system_score_codex":0.00069682044,"about_ca_system_score_gemma":0.000783419,"threshold_uncertainty_score":0.010383368},"labels":[],"label_agreement":null},{"id":"W4285199588","doi":"10.1007/978-3-031-79167-3_1","title":"Background and Definitions","year":2022,"lang":"en","type":"book-chapter","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Canadian Institute for Advanced Research","funders":"","keywords":"Computer science","score_opus":0.09273344441866999,"score_gpt":0.25646224030904613,"score_spread":0.16372879589037614,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285199588","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00091405277,0.023896221,0.0767995,0.0046594357,0.0026821692,0.00017119118,0.0015564554,0.0004275869,0.8888934],"genre_scores_gemma":[0.029651554,0.043215107,0.06276114,0.0047584153,0.004898174,0.0010177889,0.0042008604,0.0009652396,0.8485318],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9992843,0.00015924519,0.000045272605,0.00019168908,0.00024525428,0.00007418206],"domain_scores_gemma":[0.9993729,0.0002772458,0.00003624766,0.00008431283,0.0001786282,0.000050744515],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008906556,0.0016315122,0.0009793536,0.0033402788,0.0022165142,0.0041627083,0.0020507094,0.0016748641,0.09649232],"category_scores_gemma":[0.002513404,0.0006353153,0.0005834514,0.0050908793,0.002743212,0.006478238,0.0020374565,0.0037638117,0.06553008],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000009622554,0.000034719927,0.00005194249,0.00028769273,0.0000023933935,0.00004157179,0.00022862415,0.00035011483,0.00016991493,0.80561453,0.12511843,0.068090394],"study_design_scores_gemma":[0.0000020192033,0.000009849285,0.00006115469,0.00017266435,0.0000027829126,0.000083296356,0.00008033767,0.0003148115,0.00009770285,0.26397488,0.73519236,0.0000080755035],"about_ca_topic_score_codex":0.002763004,"about_ca_topic_score_gemma":0.0030633658,"teacher_disagreement_score":0.09649232,"about_ca_system_score_codex":0.001753367,"about_ca_system_score_gemma":0.0016295577,"threshold_uncertainty_score":0.32279897},"labels":[],"label_agreement":null},{"id":"W4285255083","doi":"10.1007/978-3-031-79167-3_2","title":"Reinforcement Learning Theory","year":2022,"lang":"en","type":"book-chapter","venue":"Synthesis lectures on artificial intelligence and machine learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Canadian Institute for Advanced Research","funders":"","keywords":"Reinforcement learning; Reinforcement; Computer science; Plan (archaeology); Context (archaeology); Cognitive science; Artificial intelligence; Psychology; Social psychology; History","score_opus":0.04072940189197726,"score_gpt":0.2609339799200801,"score_spread":0.22020457802810284,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285255083","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0029739572,0.01712814,0.43083987,0.005412612,0.0015944602,0.00006378479,0.0002706502,0.00045266372,0.5412639],"genre_scores_gemma":[0.324924,0.017867804,0.08920492,0.0027188705,0.0018195567,0.0003924222,0.000493066,0.00038354128,0.56219584],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9996815,0.000097587756,0.00000928932,0.00006117237,0.00012487189,0.000025578187],"domain_scores_gemma":[0.999686,0.00018079836,0.000015792757,0.00004199346,0.000057975245,0.000017431528],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004618171,0.00089751254,0.0007600316,0.0005182783,0.00047297255,0.0015512472,0.0008542576,0.0011381129,0.024864402],"category_scores_gemma":[0.0015124694,0.00030395738,0.00039572432,0.00058416696,0.0015826019,0.0014138988,0.00067660783,0.001948968,0.0053499686],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000009482601,0.000025356241,0.0000700151,0.00008206214,0.000015048729,0.000026068587,0.00005077343,0.01363208,0.00024085912,0.8889941,0.030582465,0.06627173],"study_design_scores_gemma":[0.0000093105755,0.000013312284,0.00009234592,0.00005366662,0.0000077800405,0.000036587615,0.000017354942,0.02038667,0.00023452779,0.8993231,0.0798174,0.000008035408],"about_ca_topic_score_codex":0.0016429878,"about_ca_topic_score_gemma":0.0014391869,"teacher_disagreement_score":0.024864402,"about_ca_system_score_codex":0.0014257928,"about_ca_system_score_gemma":0.0007569863,"threshold_uncertainty_score":0.08317971},"labels":[],"label_agreement":null},{"id":"W4285600734","doi":"10.24963/ijcai.2022/478","title":"CCLF: A Contrastive-Curiosity-Driven Learning Framework for Sample-Efficient Reinforcement Learning","year":2022,"lang":"en","type":"article","venue":"Proceedings of the Thirty-First International Joint Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"BC Research (Canada)","funders":"Nanyang Technological University; National Research Foundation Singapore; National Research Foundation","keywords":"Computer science; Reinforcement learning; Curiosity; Artificial intelligence; Encoder; Exploit; Sample (material); Machine learning; Code (set theory); Encoding (memory); Representation (politics)","score_opus":0.06430454282604245,"score_gpt":0.29373130230923883,"score_spread":0.22942675948319638,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285600734","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007374036,0.00027393023,0.9896254,0.00024915836,0.00003898202,0.00011317956,0.00006554285,0.0008719119,0.0013879024],"genre_scores_gemma":[0.60264546,0.00026940892,0.3917251,0.00060829596,0.000109575274,0.00076384156,0.00028880817,0.0003269286,0.0032626395],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99906594,0.0003692004,0.00004480981,0.00019261942,0.00020610017,0.00012128721],"domain_scores_gemma":[0.9974492,0.0015185266,0.00024299542,0.00023039432,0.00037291023,0.0001858608],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029639516,0.0015701626,0.0015105932,0.0007613683,0.00045710147,0.0010254126,0.0034074378,0.0017863153,0.003992889],"category_scores_gemma":[0.008099501,0.0006680427,0.0007215178,0.00048398512,0.0018359326,0.0012926112,0.0021457328,0.0029027436,0.0006958628],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001487363,0.00021260716,0.0013945759,0.00017697844,0.0000909873,0.00010124457,0.00013251926,0.85408837,0.0027607682,0.021965789,0.0033449158,0.11558258],"study_design_scores_gemma":[0.0000211617,0.00004608157,0.000051813422,0.000009630488,0.000005082447,0.00001115088,0.0000036869985,0.99271446,0.00033822496,0.006336604,0.00045615694,0.0000059674917],"about_ca_topic_score_codex":0.0042967717,"about_ca_topic_score_gemma":0.0054123034,"teacher_disagreement_score":0.0042967717,"about_ca_system_score_codex":0.0015095077,"about_ca_system_score_gemma":0.0021707613,"threshold_uncertainty_score":0.015675008},"labels":[],"label_agreement":null},{"id":"W4285604500","doi":"10.24963/ijcai.2022/440","title":"Multi-policy Grounding and Ensemble Policy Learning for Transfer Learning with Dynamics Mismatch","year":2022,"lang":"en","type":"article","venue":"Proceedings of the Thirty-First International Joint Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; University of Toronto","funders":"","keywords":"Computer science; Task (project management); Quality (philosophy); Imitation; Feature (linguistics); Policy learning; Artificial intelligence; Transfer of learning; Ensemble learning; Machine learning; Algorithm; Engineering","score_opus":0.055653521465682323,"score_gpt":0.28927776346081047,"score_spread":0.23362424199512816,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285604500","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013246997,0.00016600409,0.9850551,0.00011669704,0.00003705635,0.000032587886,0.000015677006,0.00043619704,0.00089359644],"genre_scores_gemma":[0.7903178,0.0001882864,0.20579298,0.00022118191,0.00006949922,0.00024050204,0.00013120638,0.00015045813,0.002888124],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993881,0.00017611643,0.000037046713,0.00015818713,0.00013290746,0.00010762046],"domain_scores_gemma":[0.99790144,0.0012658151,0.00021884542,0.0002256922,0.0002636647,0.00012451885],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017494113,0.0011657036,0.0016808914,0.00063984026,0.00059878285,0.0010195387,0.0018550695,0.0020745303,0.0027211446],"category_scores_gemma":[0.0061750086,0.00064580585,0.0006900317,0.00062246836,0.0012359475,0.0019331694,0.0019548193,0.0023151108,0.0005068091],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006422516,0.00005565157,0.00054698583,0.00003197197,0.00003521807,0.000055087938,0.000062929495,0.92854106,0.0011096739,0.0071988683,0.00050156435,0.061796788],"study_design_scores_gemma":[0.0000041319895,0.0000173757,0.0000247633,0.000002788569,0.000002245729,0.0000052700625,0.0000026341056,0.99782795,0.00021750729,0.0017835519,0.0001096328,0.000002258575],"about_ca_topic_score_codex":0.0044042906,"about_ca_topic_score_gemma":0.0022595753,"teacher_disagreement_score":0.0044042906,"about_ca_system_score_codex":0.0009915106,"about_ca_system_score_gemma":0.0016996098,"threshold_uncertainty_score":0.009251893},"labels":[],"label_agreement":null},{"id":"W4286910437","doi":"10.48550/arxiv.2110.02355","title":"Robustness and sample complexity of model-based MARL for general-sum Markov games","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Ministère de la Défense Nationale","keywords":"Markov chain; Mathematics; Markov kernel; Markov process; Markov perfect equilibrium; Discrete mathematics; Markov model; Applied mathematics; Combinatorics; Mathematical optimization; Variable-order Markov model; Nash equilibrium; Statistics","score_opus":0.12315007179665002,"score_gpt":0.21884184029032583,"score_spread":0.0956917684936758,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4286910437","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10223618,0.0009359966,0.88547814,0.0023019444,0.00009099054,0.00023268955,0.00028931684,0.00067771145,0.007757117],"genre_scores_gemma":[0.9351048,0.0005547055,0.059813116,0.00059825263,0.00010025589,0.0004237617,0.00038927482,0.00021624194,0.0027996372],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9942678,0.0026987344,0.00026483327,0.001176576,0.000893609,0.0006984334],"domain_scores_gemma":[0.90556556,0.08215713,0.0045340145,0.0037630717,0.0022203648,0.0017598434],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01258505,0.0018245043,0.002942761,0.0013050112,0.0011299746,0.00268017,0.003206762,0.0024406884,0.0034104008],"category_scores_gemma":[0.0671374,0.0011149086,0.00139483,0.00061229226,0.0040898775,0.004634951,0.00456208,0.005294685,0.0005204498],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034499427,0.00011206768,0.0012838695,0.00015410762,0.00008657657,0.00006189673,0.0000839867,0.9376976,0.0008320217,0.049824942,0.0007328832,0.008785136],"study_design_scores_gemma":[0.000014396062,0.00003616106,0.00008846033,0.000015414307,0.00000617871,0.0000096220265,0.000008522303,0.98100716,0.00021859049,0.018510275,0.000077545155,0.0000077785535],"about_ca_topic_score_codex":0.0049096835,"about_ca_topic_score_gemma":0.0040585278,"teacher_disagreement_score":0.01258505,"about_ca_system_score_codex":0.0042300606,"about_ca_system_score_gemma":0.0042825737,"threshold_uncertainty_score":0.06655687},"labels":[],"label_agreement":null},{"id":"W4287251902","doi":"10.48550/arxiv.2103.15793","title":"LASER: Learning a Latent Action Space for Efficient Reinforcement\\n Learning","year":2021,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Reinforcement learning; Action (physics); Computer science; Artificial intelligence; Space (punctuation); Task (project management); Machine learning; Engineering; Physics","score_opus":0.0890968772339231,"score_gpt":0.21438475174161317,"score_spread":0.12528787450769008,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4287251902","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007333932,0.00013026304,0.98972553,0.00016441153,0.000025464979,0.000045837794,0.00007798007,0.0016444414,0.0008520964],"genre_scores_gemma":[0.49320605,0.00020603232,0.49987814,0.0003558899,0.000056049015,0.0005585207,0.0005654481,0.00060915906,0.0045647672],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993814,0.00022165962,0.0000268859,0.00015772361,0.0001418405,0.000070369046],"domain_scores_gemma":[0.9987783,0.0007328247,0.000104406055,0.00016912565,0.000112050875,0.00010326864],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014216298,0.00092448737,0.0009235488,0.00042847975,0.00041047743,0.0009205035,0.0020308064,0.0014837761,0.005121836],"category_scores_gemma":[0.0051761367,0.0007280501,0.0007098967,0.0003791965,0.0013195612,0.0014138429,0.0022258435,0.002883279,0.0011773665],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015863091,0.000116428775,0.0009181312,0.00012436889,0.00006228341,0.00008867213,0.00012210722,0.85947275,0.0050088223,0.017257087,0.0034757142,0.11319499],"study_design_scores_gemma":[0.000012077539,0.000017941673,0.0000379066,0.0000048414367,0.0000021911612,0.0000065562567,0.0000039953834,0.99486285,0.0005725186,0.0041365144,0.00033917098,0.000003489964],"about_ca_topic_score_codex":0.006733193,"about_ca_topic_score_gemma":0.009060289,"teacher_disagreement_score":0.006733193,"about_ca_system_score_codex":0.0011273565,"about_ca_system_score_gemma":0.0020109469,"threshold_uncertainty_score":0.01713419},"labels":[],"label_agreement":null},{"id":"W4287333197","doi":"10.48550/arxiv.2102.02639","title":"Improving Reinforcement Learning with Human Assistance: An Argument for\\n Human Subject Studies with HIPPO Gym","year":2021,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Northern Alberta Institute of Technology","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Human–computer interaction; Subject (documents); World Wide Web","score_opus":0.09487229354682383,"score_gpt":0.2386730474669452,"score_spread":0.14380075392012137,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4287333197","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022825161,0.0017838927,0.88797235,0.032322425,0.00038574034,0.00014374802,0.000093492694,0.002965704,0.051507488],"genre_scores_gemma":[0.75535595,0.0012643564,0.21680352,0.0051204995,0.00046377644,0.00047032323,0.00012278817,0.00074709556,0.01965171],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9952709,0.0028157118,0.000085409025,0.0007558872,0.0008685067,0.0002035647],"domain_scores_gemma":[0.9791542,0.015384238,0.00079694786,0.0026887325,0.0011036104,0.0008722129],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00801839,0.00093314395,0.0006573645,0.0004944145,0.000904389,0.002034174,0.0021552735,0.0022212225,0.012370285],"category_scores_gemma":[0.026896322,0.00038707498,0.00061103364,0.0004549159,0.0065293843,0.0059181172,0.0037614056,0.004346148,0.0019446681],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007511994,0.00067988934,0.003557609,0.0006704561,0.00017265938,0.00016590673,0.0017065803,0.07736798,0.0032780294,0.63127196,0.02236606,0.25801164],"study_design_scores_gemma":[0.0005203006,0.00069093826,0.0022749756,0.00027251287,0.00007744449,0.00015735808,0.00038053517,0.25109333,0.0068957563,0.62771827,0.1098426,0.00007598658],"about_ca_topic_score_codex":0.0035242038,"about_ca_topic_score_gemma":0.002493668,"teacher_disagreement_score":0.012370285,"about_ca_system_score_codex":0.0019849741,"about_ca_system_score_gemma":0.0025521745,"threshold_uncertainty_score":0.042405784},"labels":[],"label_agreement":null},{"id":"W4287647350","doi":"10.48550/arxiv.2010.01062","title":"Exploration in Approximate Hyper-State Space for Meta Reinforcement\\n Learning","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Engineering and Physical Sciences Research Council; European Commission; Nvidia; Compute Canada; Microsoft Research","keywords":"Reinforcement learning; Task (project management); Meta learning (computer science); Computer science; State space; Artificial intelligence; Space (punctuation); Machine learning; State (computer science); Mathematics; Engineering; Algorithm","score_opus":0.17159142916653772,"score_gpt":0.2178853611837374,"score_spread":0.046293932017199696,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4287647350","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.061306644,0.000567116,0.932872,0.00063068286,0.000054098506,0.00006814799,0.000098561904,0.0007857018,0.0036171074],"genre_scores_gemma":[0.9107908,0.00021397632,0.08560833,0.00024987396,0.000042482967,0.00025369925,0.00014587743,0.000095343225,0.0025996836],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993968,0.00029911828,0.00003342914,0.0001150384,0.00008710418,0.00006845165],"domain_scores_gemma":[0.997542,0.0016111351,0.00022902832,0.00028581455,0.00017250974,0.0001594125],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014022579,0.0009447442,0.0012383166,0.0005384186,0.0003724956,0.0010484842,0.0014627449,0.0013221451,0.0029177554],"category_scores_gemma":[0.0064693713,0.000535076,0.00058077055,0.000458595,0.0016485027,0.0019246212,0.0020018085,0.0019223592,0.0004374295],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011488999,0.000061289444,0.0010151493,0.000062823354,0.000046722384,0.000054947257,0.00007489043,0.952856,0.00093898986,0.020345574,0.00083407393,0.023594566],"study_design_scores_gemma":[0.000009105443,0.0000230805,0.000044359993,0.0000064344677,0.0000036202684,0.0000071374566,0.000005113515,0.9906549,0.00015426811,0.008911991,0.00017675695,0.0000032254102],"about_ca_topic_score_codex":0.002069753,"about_ca_topic_score_gemma":0.0025285701,"teacher_disagreement_score":0.0029177554,"about_ca_system_score_codex":0.001168576,"about_ca_system_score_gemma":0.0010379549,"threshold_uncertainty_score":0.009760916},"labels":[],"label_agreement":null},{"id":"W4288047783","doi":"10.1109/icuas54217.2022.9836052","title":"Soft Actor-Critic with Inhibitory Networks for Retraining UAV Controllers Faster","year":2022,"lang":"en","type":"article","venue":"2022 International Conference on Unmanned Aircraft Systems (ICUAS)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Lockheed Martin (Canada)","funders":"","keywords":"Computer science; Retraining; Distributed computing; Control engineering; Engineering; Business","score_opus":0.030338078660022628,"score_gpt":0.25649256579083246,"score_spread":0.22615448713080982,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4288047783","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044010203,0.0005886922,0.9485874,0.0003130229,0.00013854005,0.00007577857,0.000042687156,0.0015162039,0.0047275866],"genre_scores_gemma":[0.9494313,0.00014411003,0.046493594,0.00020940948,0.000046068504,0.000088756184,0.00007012694,0.00008523105,0.0034314245],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99958163,0.000111010115,0.000023342753,0.000099607285,0.00011817315,0.00006623131],"domain_scores_gemma":[0.9988845,0.00062477053,0.00013170506,0.00008084784,0.00020238048,0.00007571434],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012749643,0.0013987349,0.00085590186,0.00041264607,0.00032082025,0.00075358833,0.00114485,0.00087029446,0.0016129746],"category_scores_gemma":[0.003915837,0.00044439078,0.00036660756,0.00022682741,0.00085486914,0.00070784095,0.000834687,0.001632402,0.00036664284],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008102531,0.000039307106,0.00041657055,0.000039065504,0.00003053796,0.000060676706,0.000029918387,0.9679232,0.0021041888,0.0020626578,0.0005877245,0.026625134],"study_design_scores_gemma":[0.000003149147,0.00001040281,0.000027928261,0.0000024209562,0.0000030771303,0.0000042597558,0.0000015638473,0.9991059,0.00032402805,0.0004313964,0.00008442505,0.0000015460024],"about_ca_topic_score_codex":0.0065653007,"about_ca_topic_score_gemma":0.007202276,"teacher_disagreement_score":0.0065653007,"about_ca_system_score_codex":0.00086437265,"about_ca_system_score_gemma":0.0010904606,"threshold_uncertainty_score":0.013054192},"labels":[],"label_agreement":null},{"id":"W4289829023","doi":"10.1109/med54222.2022.9837194","title":"Time-delayed Data Transmission in Heterogeneous Multi-agent Deep Reinforcement Learning System","year":2022,"lang":"en","type":"article","venue":"2022 30th Mediterranean Conference on Control and Automation (MED)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Reinforcement learning; Computer science; Reinforcement; Transmission (telecommunications); Artificial intelligence; Data transmission; Computer network; Telecommunications; Engineering","score_opus":0.03798392676037204,"score_gpt":0.26433006271469006,"score_spread":0.22634613595431802,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4289829023","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.30011585,0.0005106824,0.693672,0.00069852953,0.00011372383,0.000057476547,0.000058294212,0.00023448898,0.0045389384],"genre_scores_gemma":[0.99474084,0.00004836614,0.004150938,0.000030175104,0.000008934992,0.000017950033,0.000009106519,0.000005044065,0.0009887349],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99932945,0.00019492554,0.00003431344,0.00014890914,0.00013448394,0.00015794572],"domain_scores_gemma":[0.9986343,0.00065716694,0.00027419976,0.0000741693,0.00023249426,0.00012776014],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013437311,0.00052700774,0.00077066617,0.00021504452,0.00046747664,0.000704209,0.001060014,0.0007720919,0.0009787808],"category_scores_gemma":[0.0028519863,0.0002929776,0.00033068762,0.00021957872,0.0009051498,0.0009275246,0.0009351251,0.0008963413,0.00009425908],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009966778,0.00003787261,0.0008888453,0.000029580518,0.000025499585,0.00015692267,0.000059924005,0.98453194,0.001146473,0.0062631606,0.00024178733,0.0065184077],"study_design_scores_gemma":[0.000007108606,0.000021100748,0.000086347216,0.000001064357,0.0000034990069,0.00000848113,0.0000056974545,0.9986619,0.00011294874,0.0010399958,0.000049615588,0.0000022629886],"about_ca_topic_score_codex":0.0071028154,"about_ca_topic_score_gemma":0.003859033,"teacher_disagreement_score":0.0071028154,"about_ca_system_score_codex":0.0012077689,"about_ca_system_score_gemma":0.0009463139,"threshold_uncertainty_score":0.014122963},"labels":[],"label_agreement":null},{"id":"W4291821302","doi":"10.32470/ccn.2022.1229-0","title":"Continual Reinforcement Learning with Multi-Timescale Successor Features","year":2022,"lang":"en","type":"article","venue":"2022 Conference on Cognitive Computational Neuroscience","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Successor cardinal; Reinforcement learning; Reinforcement; Computer science; Artificial intelligence; Engineering; Structural engineering; Mathematics","score_opus":0.0353669370020116,"score_gpt":0.2839279176125578,"score_spread":0.2485609806105462,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4291821302","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14240687,0.00038091402,0.8477125,0.0005105748,0.00020783725,0.00008606407,0.00008560378,0.0007800519,0.007829497],"genre_scores_gemma":[0.9720652,0.000055999244,0.025597233,0.000048088685,0.000026209822,0.000051228428,0.00002764995,0.000030752173,0.0020977],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996407,0.00008544232,0.000024723542,0.00009949791,0.00010161342,0.000047995578],"domain_scores_gemma":[0.99763405,0.0014512048,0.00018640138,0.00029522966,0.00022836502,0.0002048367],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012383738,0.00046081934,0.0006214749,0.00029707263,0.0002783065,0.00066300883,0.0010545825,0.00077840785,0.00475613],"category_scores_gemma":[0.0061369548,0.00032128123,0.00035672027,0.000278137,0.00064129353,0.001513341,0.0012749826,0.0017555518,0.0003767675],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008761854,0.0005604589,0.0026840435,0.00019013823,0.00011347016,0.00027271797,0.00014272511,0.6576232,0.012233692,0.05837278,0.0025566935,0.2643739],"study_design_scores_gemma":[0.00002229582,0.000065290646,0.00017208143,0.000004478534,0.0000069363828,0.000027238975,0.000003967409,0.98749304,0.0005785338,0.01139987,0.00022028576,0.0000059735394],"about_ca_topic_score_codex":0.0009484216,"about_ca_topic_score_gemma":0.001370999,"teacher_disagreement_score":0.00475613,"about_ca_system_score_codex":0.00043619197,"about_ca_system_score_gemma":0.0006334109,"threshold_uncertainty_score":0.015910804},"labels":[],"label_agreement":null},{"id":"W4293155040","doi":"","title":"Dynamic Programming in Distributional Reinforcement Learning","year":2020,"lang":"en","type":"report","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Reinforcement learning; Computer science; Reinforcement; Dynamic programming; Artificial intelligence; Psychology; Algorithm; Social psychology","score_opus":0.01766060272860948,"score_gpt":0.252790648297136,"score_spread":0.2351300455685265,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4293155040","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0099991625,0.0019832843,0.9768018,0.0018715436,0.00014616353,0.000031972715,0.00008739075,0.00017549895,0.0089032445],"genre_scores_gemma":[0.7544508,0.0038440304,0.22097015,0.0011510436,0.00069925725,0.0005148914,0.0003479185,0.00030926248,0.017712671],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99779093,0.0011694757,0.00008041825,0.00043810948,0.00035473384,0.00016638594],"domain_scores_gemma":[0.99406874,0.0047472822,0.00030413014,0.00026038598,0.00037343343,0.0002461024],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026514463,0.0010477329,0.0014120936,0.00066077046,0.0004335918,0.001794464,0.0012466956,0.0016247621,0.0045983903],"category_scores_gemma":[0.0140038105,0.0004857215,0.00071309507,0.0009475477,0.0026019586,0.0026381991,0.0021978207,0.003514169,0.0006588318],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006994484,0.00007944071,0.000589198,0.00017385371,0.00006462637,0.00008823476,0.0001125262,0.41202697,0.00039956943,0.5478915,0.002761318,0.03574288],"study_design_scores_gemma":[0.00001545346,0.000029615923,0.00009887465,0.000022064945,0.000006565021,0.000016925429,0.00001267047,0.55665237,0.00013064628,0.4412144,0.0017898355,0.000010513799],"about_ca_topic_score_codex":0.0024328101,"about_ca_topic_score_gemma":0.0017261689,"teacher_disagreement_score":0.0045983903,"about_ca_system_score_codex":0.0016711751,"about_ca_system_score_gemma":0.0013620723,"threshold_uncertainty_score":0.015383124},"labels":[],"label_agreement":null},{"id":"W4293370597","doi":"10.1109/lra.2022.3196132","title":"Safe-Control-Gym: A Unified Benchmark Suite for Safe Learning-Based Control and Reinforcement Learning in Robotics","year":2022,"lang":"en","type":"article","venue":"IEEE Robotics and Automation Letters","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":47,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dynamic Systems Analysis (Canada); Vector Institute; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Reinforcement learning; Artificial intelligence; Computer science; Suite; Benchmark (surveying); Machine learning","score_opus":0.009230860443678211,"score_gpt":0.22130848297750758,"score_spread":0.21207762253382936,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4293370597","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23418145,0.00872069,0.5439245,0.003586483,0.0021954603,0.0026135813,0.055978414,0.063825995,0.08497339],"genre_scores_gemma":[0.44783774,0.0029444287,0.4081267,0.0010873616,0.00019668585,0.0033192227,0.11590253,0.0063824076,0.014202945],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9964227,0.0012066449,0.0003572492,0.0003803386,0.0012285551,0.00040435322],"domain_scores_gemma":[0.9945082,0.0023212754,0.0003688817,0.00079399685,0.0015828733,0.0004249067],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003917353,0.003525279,0.001172202,0.0023617516,0.0007530675,0.0017862584,0.004270894,0.0018836278,0.0049179974],"category_scores_gemma":[0.01080441,0.00065889896,0.0012056172,0.0024200147,0.00089656503,0.0013884122,0.0018694558,0.0023131424,0.00175055],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000987038,0.0013754485,0.004930566,0.002274076,0.0003314166,0.00034152888,0.00016741722,0.6457854,0.004969455,0.02275209,0.14571448,0.17037107],"study_design_scores_gemma":[0.00048298927,0.0006206548,0.0020631317,0.00017343066,0.00006504506,0.00015201731,0.00010496194,0.91400796,0.010950619,0.012719887,0.05859458,0.00006472033],"about_ca_topic_score_codex":0.01607153,"about_ca_topic_score_gemma":0.0161859,"teacher_disagreement_score":0.01607153,"about_ca_system_score_codex":0.0016351647,"about_ca_system_score_gemma":0.0026700473,"threshold_uncertainty_score":0.031955957},"labels":[],"label_agreement":null},{"id":"W4293863172","doi":"10.1109/siu55565.2022.9864806","title":"Autonomous Driving Systems for Decision-Making Under Uncertainty Using Deep Reinforcement Learning","year":2022,"lang":"en","type":"article","venue":"2022 30th Signal Processing and Communications Applications Conference (SIU)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Stantec (Canada)","funders":"","keywords":"Partially observable Markov decision process; Reinforcement learning; Markov decision process; Computer science; Artificial intelligence; Action (physics); Process (computing); Observable; Control (management); Markov process; State (computer science); Autonomous agent; Markov chain; Machine learning; Markov model; Mathematics","score_opus":0.04014414957858934,"score_gpt":0.30787330434500143,"score_spread":0.26772915476641207,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4293863172","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12981069,0.00040317368,0.864063,0.0005345487,0.000075622214,0.0000711331,0.000051687595,0.00063328876,0.0043568765],"genre_scores_gemma":[0.9804638,0.00006500127,0.018395917,0.000056093944,0.0000120499235,0.000046591587,0.000035466175,0.000013905267,0.0009111591],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99974436,0.00007827081,0.00001399544,0.000056570345,0.000051459818,0.00005540723],"domain_scores_gemma":[0.99923,0.00042331222,0.00011190629,0.000044419616,0.00012703425,0.000063414445],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008888999,0.0006801148,0.0006497143,0.0002289555,0.0003484099,0.00063645595,0.0006888782,0.00070371357,0.0012415141],"category_scores_gemma":[0.0018461426,0.00033499862,0.00040160713,0.00017579754,0.0007276415,0.00053853355,0.00086821883,0.0013881836,0.0001427605],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000039034825,0.00003885388,0.00066842814,0.00002454696,0.000023635188,0.000047103673,0.000044970646,0.98162395,0.00096550514,0.0034364313,0.00026218954,0.012825347],"study_design_scores_gemma":[0.0000031458342,0.000009363787,0.00004120691,0.0000013789462,0.0000017435937,0.000002023563,0.0000020663156,0.99872893,0.00009706014,0.0010586263,0.000053314317,0.000001174486],"about_ca_topic_score_codex":0.009824483,"about_ca_topic_score_gemma":0.0076040737,"teacher_disagreement_score":0.009824483,"about_ca_system_score_codex":0.00086317013,"about_ca_system_score_gemma":0.0013999722,"threshold_uncertainty_score":0.019534588},"labels":[],"label_agreement":null},{"id":"W4293863350","doi":"10.1109/siu55565.2022.9864802","title":"Context Detection and Identification In Multi-Agent Reinforcement Learning With Non-Stationary Environment","year":2022,"lang":"en","type":"article","venue":"2022 30th Signal Processing and Communications Applications Conference (SIU)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Stantec (Canada)","funders":"","keywords":"Reinforcement learning; Identification (biology); Context (archaeology); Computer science; Artificial intelligence; Machine learning; Geography","score_opus":0.03048400050106643,"score_gpt":0.2569068097743107,"score_spread":0.22642280927324426,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4293863350","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10293014,0.0006278277,0.8936676,0.0002549743,0.00007015385,0.000089337256,0.000030726555,0.0005834127,0.0017458422],"genre_scores_gemma":[0.9478785,0.00011344423,0.050911114,0.00008546845,0.000017184273,0.00007551467,0.00002543081,0.000019549794,0.0008737837],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990816,0.0003021517,0.00005064035,0.00027420907,0.00016460886,0.00012677032],"domain_scores_gemma":[0.9981218,0.0010957231,0.00028454373,0.00012429716,0.00023306474,0.00014059724],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013766827,0.00075701444,0.001035402,0.00035259133,0.00045048795,0.0007041083,0.0010115582,0.0007517494,0.00072925346],"category_scores_gemma":[0.0041924836,0.00040063512,0.00044253137,0.00024098015,0.0008429857,0.0010398469,0.0010952073,0.0012213559,0.00011981088],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019632315,0.00015647753,0.0026658894,0.00009099,0.00006889686,0.00016127447,0.00012877752,0.920797,0.003723815,0.0061876443,0.0004604037,0.06536248],"study_design_scores_gemma":[0.00001233781,0.000034836903,0.00020030206,0.0000029101923,0.0000056325607,0.000011078237,0.0000069030666,0.9976113,0.00047226367,0.0015024684,0.00013489531,0.000005059783],"about_ca_topic_score_codex":0.0065578776,"about_ca_topic_score_gemma":0.0041739945,"teacher_disagreement_score":0.0065578776,"about_ca_system_score_codex":0.00084229634,"about_ca_system_score_gemma":0.0012297027,"threshold_uncertainty_score":0.01303941},"labels":[],"label_agreement":null},{"id":"W4294912101","doi":"10.36227/techrxiv.14842245.v2","title":"Safe Deployment of a Reinforcement Learning Robot Using Self Stabilization","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Software deployment; Robot; Robotics; Computer science; State space; Artificial intelligence; Simulation; Human–computer interaction; Software engineering; Mathematics","score_opus":0.037504982322837374,"score_gpt":0.286013155198828,"score_spread":0.24850817287599059,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4294912101","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09245086,0.00006517352,0.90246546,0.00017872051,0.00003842078,0.00010468351,0.000015455773,0.0017683447,0.002912895],"genre_scores_gemma":[0.95719355,0.00002268405,0.041168373,0.000042270804,0.0000070637157,0.00008010061,0.00001992516,0.00005813583,0.0014078576],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995074,0.00012810621,0.000025529625,0.00011516563,0.00014840723,0.00007538217],"domain_scores_gemma":[0.9984471,0.0006055076,0.00029530042,0.00025429117,0.00024579378,0.00015206417],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007685639,0.00073426735,0.0003873622,0.00026485126,0.00032965737,0.00044808615,0.0008837041,0.00078229886,0.0015001555],"category_scores_gemma":[0.0029720652,0.00027838693,0.00029010247,0.0001009799,0.0011605108,0.0005086334,0.0011878455,0.0008473041,0.00041592336],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001808686,0.00009383384,0.00077398313,0.000061896186,0.000021176225,0.00022939069,0.00016943687,0.93695176,0.023371242,0.005555195,0.00060661306,0.031984612],"study_design_scores_gemma":[0.000015849564,0.00009281741,0.00008058528,0.0000043955265,0.000002974315,0.000017024626,0.00000855798,0.9958389,0.0027926709,0.0008993208,0.00024305213,0.0000038657286],"about_ca_topic_score_codex":0.0032127071,"about_ca_topic_score_gemma":0.0021765304,"teacher_disagreement_score":0.0032127071,"about_ca_system_score_codex":0.00052736455,"about_ca_system_score_gemma":0.0009656211,"threshold_uncertainty_score":0.006388068},"labels":[],"label_agreement":null},{"id":"W4295308610","doi":"10.1109/access.2022.3206035","title":"On Meeting a Maximum Delay Constraint Using Reinforcement Learning","year":2022,"lang":"en","type":"article","venue":"IEEE Access","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Telefonaktiebolaget LM Ericsson","keywords":"Reinforcement learning; Computer science; Constraint (computer-aided design); Reinforcement; Mathematical optimization; Artificial intelligence; Mathematics; Engineering","score_opus":0.04462453885572762,"score_gpt":0.30320423508470057,"score_spread":0.2585796962289729,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4295308610","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0378845,0.00044253067,0.9575676,0.00053111376,0.00006105563,0.00005116883,0.00004873084,0.00023656346,0.0031767196],"genre_scores_gemma":[0.94127035,0.0002846065,0.05574764,0.00028115718,0.000053580934,0.00012002207,0.000080962614,0.00006031544,0.0021013175],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99941504,0.00022428622,0.000024946394,0.00010667549,0.00011812506,0.00011103407],"domain_scores_gemma":[0.99590296,0.0032502294,0.00028118512,0.00009518989,0.00031780294,0.00015268782],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016613527,0.0009121443,0.0014252939,0.0005029898,0.00042817416,0.000752128,0.0010697916,0.0011165359,0.0020229223],"category_scores_gemma":[0.00602966,0.00041620436,0.0004596979,0.0005694637,0.001329065,0.0011946389,0.0009430615,0.0015313034,0.00024111253],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000029123996,0.000021185308,0.00030311773,0.000025151794,0.0000113448705,0.000024939049,0.000015155336,0.98847663,0.00023241751,0.004794629,0.00028422129,0.005782012],"study_design_scores_gemma":[0.000004853128,0.000010275748,0.000023852943,0.0000023643845,0.000001847127,0.0000027005647,0.0000021239691,0.9979322,0.000050776718,0.0019045756,0.000063052335,0.000001335206],"about_ca_topic_score_codex":0.008928043,"about_ca_topic_score_gemma":0.006709794,"teacher_disagreement_score":0.008928043,"about_ca_system_score_codex":0.0011730302,"about_ca_system_score_gemma":0.001936972,"threshold_uncertainty_score":0.017752111},"labels":[],"label_agreement":null},{"id":"W4295683034","doi":"10.1016/j.neunet.2025.108521","title":"Multi-step first: A lightweight deep reinforcement learning strategy for robust continuous control with partial observability","year":2025,"lang":"en","type":"preprint","venue":"Neural Networks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Social Sciences and Humanities Research Council of Canada; Australian Research Council; Alliance de recherche numérique du Canada","keywords":"Observability; Robot; Computer science; State space; Robustness (evolution); Control theory (sociology); Reinforcement learning; Control (management); Control engineering; Artificial intelligence; Engineering; Mathematics","score_opus":0.03587578380555461,"score_gpt":0.2609447892771726,"score_spread":0.22506900547161796,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4295683034","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014891791,0.00015776351,0.9824765,0.00016302128,0.00003177047,0.00003326717,0.000024558463,0.00058711367,0.0016342032],"genre_scores_gemma":[0.89629495,0.00010080369,0.10093405,0.00018043305,0.000023851746,0.00010939028,0.00006224258,0.000090334,0.0022039863],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995234,0.00012653692,0.000022224074,0.000100930316,0.00014901889,0.00007775176],"domain_scores_gemma":[0.9987884,0.00065494666,0.00014505463,0.00012463795,0.00017308477,0.0001138385],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014951726,0.0010188227,0.0010688102,0.000379486,0.0003655278,0.0006777194,0.0014383298,0.0009500161,0.0022534216],"category_scores_gemma":[0.0032351927,0.00047098444,0.0005787591,0.0002802051,0.0010963258,0.0010320634,0.0017992964,0.0017104131,0.00033463442],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001450745,0.00006281915,0.00053135375,0.00008187991,0.000047081325,0.00010955802,0.000077320474,0.92240405,0.0027037757,0.01380502,0.001168123,0.058864024],"study_design_scores_gemma":[0.0000064629758,0.000030128278,0.000028628687,0.0000033510105,0.0000028077943,0.0000072443154,0.0000019388742,0.99738306,0.00030404667,0.0021002875,0.00012917118,0.0000028528211],"about_ca_topic_score_codex":0.0037415703,"about_ca_topic_score_gemma":0.004021374,"teacher_disagreement_score":0.0037415703,"about_ca_system_score_codex":0.00090990606,"about_ca_system_score_gemma":0.0014490845,"threshold_uncertainty_score":0.007907271},"labels":[],"label_agreement":null},{"id":"W4296182512","doi":"10.31234/osf.io/z8yrv","title":"Action chunking as conditional policy compression","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Office of Naval Research; National Science Foundation","keywords":"Chunking (psychology); Action (physics); Compression (physics); Computer science; Econometrics; Natural language processing; Artificial intelligence; Economics","score_opus":0.055071456781607964,"score_gpt":0.3614701003934595,"score_spread":0.30639864361185154,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4296182512","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16335808,0.00065911206,0.82239056,0.0012441747,0.00012712675,0.00015126087,0.00038972672,0.0011716364,0.010508392],"genre_scores_gemma":[0.9175553,0.0003157295,0.07779314,0.00023489565,0.00007101373,0.00019402201,0.00023178518,0.00012678633,0.003477303],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99927,0.00017980665,0.00005021612,0.00021988203,0.00018182556,0.00009819516],"domain_scores_gemma":[0.9932707,0.0040293788,0.00086181203,0.0011173995,0.0004309578,0.00028978396],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008971201,0.00065761793,0.0006419026,0.00052060804,0.00033929356,0.0010707709,0.0011292042,0.0008363963,0.0044632223],"category_scores_gemma":[0.010244579,0.00044545296,0.0004946445,0.0004703393,0.0017574742,0.0033307518,0.0013532502,0.0014477693,0.00038174764],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063844566,0.00028078994,0.004736216,0.0003448473,0.0001544352,0.00051774504,0.00065538986,0.6408086,0.02563858,0.19550063,0.002600694,0.12812364],"study_design_scores_gemma":[0.000030572715,0.00009308829,0.0016982947,0.00003086161,0.000033165452,0.000109291585,0.000033327535,0.840206,0.0054113325,0.15072477,0.0016013612,0.000027876515],"about_ca_topic_score_codex":0.0032919906,"about_ca_topic_score_gemma":0.0019997433,"teacher_disagreement_score":0.0044632223,"about_ca_system_score_codex":0.001273699,"about_ca_system_score_gemma":0.0012161857,"threshold_uncertainty_score":0.0149309635},"labels":[],"label_agreement":null},{"id":"W4296474750","doi":"10.1109/cog51982.2022.9893546","title":"Mitigating Cowardice for Reinforcement Learning Agents in Combat Scenarios","year":2022,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Cowardice; Reinforcement learning; Computer science; Reinforcement; Computer security; Artificial intelligence; Engineering; Geography","score_opus":0.037629556357336454,"score_gpt":0.28277255263449574,"score_spread":0.2451429962771593,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4296474750","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24541879,0.0003460296,0.74944943,0.00049933087,0.000064918095,0.00014937078,0.000020730831,0.0010177789,0.0030335654],"genre_scores_gemma":[0.96918553,0.000054076383,0.02983018,0.00010686691,0.000011638052,0.000047646838,0.000014713152,0.000026825257,0.0007224381],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991487,0.00036591484,0.000050846815,0.00012461552,0.00016420645,0.00014561946],"domain_scores_gemma":[0.9951644,0.003111118,0.00066402165,0.0003057771,0.00045893472,0.00029566188],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024872075,0.0011326186,0.000996668,0.0004303917,0.00045012071,0.0007402298,0.0012108139,0.0010747638,0.0011623678],"category_scores_gemma":[0.008348693,0.00043843035,0.00032927762,0.0001549435,0.0012881873,0.0011580167,0.0012038936,0.0019353384,0.0001855664],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024713489,0.00022572171,0.0025698638,0.00008167789,0.00006604171,0.00008796216,0.00012093691,0.9458408,0.0051717344,0.0041494966,0.00047709586,0.04096151],"study_design_scores_gemma":[0.00002088587,0.00012963824,0.00021837861,0.0000065348,0.000008137485,0.000018109671,0.000011027987,0.9968266,0.0009979559,0.0015616978,0.00019493214,0.0000060166103],"about_ca_topic_score_codex":0.0036881804,"about_ca_topic_score_gemma":0.003791045,"teacher_disagreement_score":0.0036881804,"about_ca_system_score_codex":0.00087125355,"about_ca_system_score_gemma":0.0012806711,"threshold_uncertainty_score":0.013153791},"labels":[],"label_agreement":null},{"id":"W4297822508","doi":"10.48550/arxiv.2103.02142","title":"Learning to Fly -- a Gym Environment with PyBullet Physics for\\n Reinforcement Learning of Multi-agent Quadcopter Control","year":2021,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; Institute for Christian Studies; University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Software portability; Artificial intelligence; Human–computer interaction; Physics engine; Machine learning; Operating system","score_opus":0.05253152417695468,"score_gpt":0.18862842276405678,"score_spread":0.1360968985871021,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4297822508","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015620668,0.00018322718,0.8913734,0.0007299128,0.00016735643,0.00024685508,0.0014965523,0.07268029,0.017501788],"genre_scores_gemma":[0.26642215,0.00033541297,0.7029899,0.0005016895,0.00006805811,0.0009692926,0.0025138287,0.008464671,0.017735025],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997501,0.00004255064,0.000016750844,0.00006349683,0.00009458917,0.000032465545],"domain_scores_gemma":[0.99962187,0.00014839381,0.000023421719,0.000084503146,0.000029079998,0.00009266544],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006459296,0.00068098446,0.0005964331,0.0003476173,0.00058855966,0.000865061,0.002004101,0.00093446963,0.033020444],"category_scores_gemma":[0.0017692727,0.0005511469,0.0010655877,0.000263785,0.0009858002,0.0019516671,0.002717335,0.0024088041,0.006478161],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010818598,0.0009276001,0.0041643535,0.000705868,0.00014786558,0.0009535249,0.000756547,0.3986495,0.035579152,0.13047527,0.07830443,0.348254],"study_design_scores_gemma":[0.00027715936,0.00017128738,0.0015124697,0.000068987654,0.000018225666,0.00019177461,0.000048242233,0.7819519,0.012559183,0.06205875,0.14107153,0.000070459035],"about_ca_topic_score_codex":0.0022038375,"about_ca_topic_score_gemma":0.0035925952,"teacher_disagreement_score":0.033020444,"about_ca_system_score_codex":0.0005642841,"about_ca_system_score_gemma":0.0013952379,"threshold_uncertainty_score":0.110464394},"labels":[],"label_agreement":null},{"id":"W4297848563","doi":"10.48550/arxiv.2001.06627","title":"Multi-agent Motion Planning for Dense and Dynamic Environments via Deep\\n Reinforcement Learning","year":2020,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Toronto Metropolitan University","funders":"","keywords":"Reinforcement learning; Leverage (statistics); Computer science; Motion planning; Mathematical optimization; Artificial intelligence; Collision; Path (computing); Simple (philosophy); Algorithm; Mathematics; Robot","score_opus":0.0764126386328068,"score_gpt":0.2122913612488068,"score_spread":0.13587872261599998,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4297848563","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018330643,0.00012594285,0.9791899,0.00015023645,0.000027667294,0.000029413533,0.00001634849,0.00047863644,0.0016513264],"genre_scores_gemma":[0.8114507,0.00013112088,0.18540563,0.00015698143,0.000028093731,0.00012646287,0.00007588108,0.00006181426,0.0025633227],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99974424,0.000056459743,0.000012783915,0.00006368568,0.00007057986,0.000052284042],"domain_scores_gemma":[0.9994692,0.0002626759,0.00007474264,0.000056955847,0.00007806733,0.000058360925],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006472406,0.0007465435,0.0007084946,0.00036470895,0.00042325445,0.00052097545,0.0013302282,0.0008819262,0.0012321937],"category_scores_gemma":[0.0014404667,0.0004662519,0.0005160519,0.0002928539,0.0009519201,0.0007372603,0.001203701,0.0011581712,0.00022268776],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000026459815,0.00002811102,0.0004121212,0.000019804054,0.00001953277,0.0000406523,0.000029241655,0.96667564,0.0010624774,0.003832378,0.00040726102,0.027446447],"study_design_scores_gemma":[0.0000039831184,0.0000072667135,0.000023089011,0.0000012657786,0.0000013363918,0.0000037521668,0.0000018884091,0.99872714,0.0001357472,0.0009777211,0.00011549494,0.000001196233],"about_ca_topic_score_codex":0.01077933,"about_ca_topic_score_gemma":0.010301871,"teacher_disagreement_score":0.01077933,"about_ca_system_score_codex":0.00092882366,"about_ca_system_score_gemma":0.0012927264,"threshold_uncertainty_score":0.021433175},"labels":[],"label_agreement":null},{"id":"W4298364546","doi":"","title":"A robot to study the development of artwork appreciation through social interactions","year":2013,"lang":"en","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"NeuroDevNet","funders":"","keywords":"Robot; Human–computer interaction; Computer science; Artificial intelligence","score_opus":0.027332943354547532,"score_gpt":0.2639089202192305,"score_spread":0.23657597686468299,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4298364546","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6826664,0.0003463931,0.27363515,0.0010127062,0.00017881635,0.00037234972,0.00015116608,0.0008089366,0.04082813],"genre_scores_gemma":[0.88721865,0.000108344306,0.10366415,0.000078955854,0.000014831589,0.00021010607,0.00006175299,0.000052309715,0.00859091],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9998173,0.00009604186,0.0000033196986,0.000042753105,0.000024373649,0.000016297612],"domain_scores_gemma":[0.99948645,0.0002888763,0.00003969355,0.0000568599,0.00003251095,0.000095619274],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000312429,0.00032878763,0.00021192375,0.00020114738,0.0005345889,0.0007626808,0.00048824976,0.00070906803,0.0066039898],"category_scores_gemma":[0.001401866,0.00018391675,0.0003282921,0.00011662093,0.0010893838,0.0007867932,0.0011232819,0.0005792066,0.00055285316],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001874271,0.002255473,0.024106715,0.0011347323,0.00021610613,0.0027929908,0.019673264,0.07200835,0.46684808,0.12119151,0.009546551,0.27835193],"study_design_scores_gemma":[0.00079458015,0.0049627223,0.042741347,0.00021559263,0.0002085005,0.002305634,0.012548458,0.69258046,0.08534269,0.07816193,0.07988751,0.00025059152],"about_ca_topic_score_codex":0.00053550943,"about_ca_topic_score_gemma":0.0005903015,"teacher_disagreement_score":0.0066039898,"about_ca_system_score_codex":0.00020556946,"about_ca_system_score_gemma":0.00032515457,"threshold_uncertainty_score":0.022092521},"labels":[],"label_agreement":null},{"id":"W4300126582","doi":"10.48550/arxiv.2005.06223","title":"DREAM Architecture: a Developmental Approach to Open-Ended Learning in\\n Robotics","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Animal Health Institute","funders":"","keywords":"Reinforcement learning; Artificial intelligence; Computer science; Robot; Flexibility (engineering); Task (project management); Adaptation (eye); Robotics; Cognitive architecture; Architecture; Robot learning; Restructuring; Behavior-based robotics; Scale (ratio); Human–computer interaction; Cognition; Engineering; Mobile robot; Psychology; Mathematics","score_opus":0.09995570773760962,"score_gpt":0.20679875872831407,"score_spread":0.10684305099070446,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4300126582","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0063012787,0.00022283944,0.9767667,0.00066566723,0.00003946117,0.000090318565,0.000053779237,0.0005627291,0.01529727],"genre_scores_gemma":[0.20659816,0.00044338166,0.77800155,0.00027188164,0.000027562282,0.00042848432,0.00016977433,0.00019587063,0.013863311],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990232,0.0004122741,0.00005672974,0.00024680846,0.00019435746,0.000066669214],"domain_scores_gemma":[0.99842787,0.0007000833,0.000105458,0.00036892024,0.00022381054,0.00017393254],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018657261,0.00052775803,0.00036661592,0.0005699954,0.0006777905,0.002263328,0.0034031398,0.0013575281,0.0056120055],"category_scores_gemma":[0.0057539195,0.0005657325,0.0008286905,0.00035388148,0.004069193,0.0032228976,0.003565088,0.0026701195,0.0010475552],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000056556888,0.00009515235,0.00086859614,0.00023388542,0.000035032175,0.00014209346,0.001597247,0.052456867,0.0029546102,0.8385027,0.002101127,0.100956105],"study_design_scores_gemma":[0.00003336997,0.00012235902,0.00034036755,0.00013391099,0.000024971325,0.0001741871,0.0003476948,0.2898093,0.0054562264,0.66146016,0.0420555,0.000041972362],"about_ca_topic_score_codex":0.0024260634,"about_ca_topic_score_gemma":0.0037144616,"teacher_disagreement_score":0.0056120055,"about_ca_system_score_codex":0.001572609,"about_ca_system_score_gemma":0.0017289484,"threshold_uncertainty_score":0.018774092},"labels":[],"label_agreement":null},{"id":"W4300362000","doi":"10.48550/arxiv.1712.02441","title":"A Novel Model for Arbitration between Planning and Habitual Control\\n Systems","year":2017,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Action selection; Reinforcement learning; Internal model; Task (project management); Control (management); Action (physics); Artificial intelligence; Kinematics; A priori and a posteriori; Machine learning; Human–computer interaction; Engineering","score_opus":0.15153388522042371,"score_gpt":0.22919272519554015,"score_spread":0.07765883997511644,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4300362000","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028174132,0.00030810724,0.9528049,0.0007523543,0.00013007496,0.00006743136,0.00025588457,0.0012880847,0.016218957],"genre_scores_gemma":[0.883945,0.00039476456,0.08993759,0.00024227863,0.000104847386,0.00033854172,0.00027435154,0.00015874252,0.024603916],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996754,0.000048558853,0.00001800041,0.00012429229,0.00007028699,0.000063425214],"domain_scores_gemma":[0.9996662,0.00011407729,0.000051065537,0.000060252198,0.00006204546,0.000046221332],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004531952,0.000630943,0.00065766054,0.0002803654,0.0004858963,0.0013919399,0.0019921134,0.0012220575,0.006778385],"category_scores_gemma":[0.0010498876,0.00042116377,0.0007975096,0.00029551925,0.0011428919,0.001904178,0.0014714431,0.0020678786,0.00088256766],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015124024,0.00008484148,0.00073593727,0.00011458226,0.0000671056,0.00025238132,0.00018297067,0.7699502,0.010182542,0.18475288,0.0026595418,0.030865734],"study_design_scores_gemma":[0.000018339993,0.00003207098,0.00007863863,0.000003824262,0.00000889299,0.000028864995,0.000005088715,0.97484905,0.0005423479,0.022684097,0.0017417633,0.000007005096],"about_ca_topic_score_codex":0.0038163823,"about_ca_topic_score_gemma":0.0040052454,"teacher_disagreement_score":0.006778385,"about_ca_system_score_codex":0.0009342026,"about_ca_system_score_gemma":0.0013499408,"threshold_uncertainty_score":0.022675991},"labels":[],"label_agreement":null},{"id":"W4301808900","doi":"10.48550/arxiv.2011.04118","title":"Joint Estimation of Expertise and Reward Preferences From Human Demonstrations","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Inference; Leverage (statistics); Robot; Computer science; Set (abstract data type); Artificial intelligence; Function (biology); Human–robot interaction; Machine learning; Human–computer interaction; Space (punctuation)","score_opus":0.14989913086949452,"score_gpt":0.21265690373065238,"score_spread":0.06275777286115786,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4301808900","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.35594046,0.0003384646,0.6399477,0.00048019696,0.000016901948,0.00008008245,0.00017868221,0.00051203644,0.0025054673],"genre_scores_gemma":[0.96828896,0.000057697536,0.030880399,0.000041097403,0.000008646945,0.000028703527,0.00008605874,0.000018682142,0.0005897127],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988182,0.00058307813,0.000049974515,0.00027985414,0.00017728844,0.00009146616],"domain_scores_gemma":[0.9908414,0.0066720736,0.00094252697,0.00064464106,0.0005034222,0.00039586768],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025260695,0.0005527742,0.0008677945,0.00056376617,0.00020367735,0.0006890367,0.0008026077,0.0009797777,0.0016605092],"category_scores_gemma":[0.01885344,0.00044344645,0.00041803904,0.00028507973,0.00093461835,0.0014090711,0.0009821015,0.0013144454,0.0002921461],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007617306,0.00029657112,0.026941026,0.0002573035,0.00018260903,0.0003517375,0.00046015604,0.82461095,0.007666525,0.0071700057,0.001224977,0.1300765],"study_design_scores_gemma":[0.000027101578,0.00010738762,0.006161517,0.000016778491,0.000013006976,0.00007998001,0.00004061551,0.98257935,0.0017994627,0.008932077,0.00022377894,0.00001892919],"about_ca_topic_score_codex":0.0027460048,"about_ca_topic_score_gemma":0.0035742382,"teacher_disagreement_score":0.0027460048,"about_ca_system_score_codex":0.00063357,"about_ca_system_score_gemma":0.0006151168,"threshold_uncertainty_score":0.013359308},"labels":[],"label_agreement":null},{"id":"W4302917494","doi":"10.48550/arxiv.1806.06161","title":"BaRC: Backward Reachability Curriculum for Robotic Reinforcement\\n Learning","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Office of Naval Research; Natural Sciences and Engineering Research Council of Canada; Toyota Research Institute","keywords":"Reinforcement learning; Computer science; Leverage (statistics); Bottleneck; Reachability; Prior probability; Flexibility (engineering); Artificial intelligence; Optimal control; Curriculum; Mathematical optimization; Machine learning; Bayesian probability; Theoretical computer science; Mathematics","score_opus":0.06339507954544266,"score_gpt":0.21067285721099221,"score_spread":0.14727777766554956,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4302917494","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012885177,0.00011713192,0.9807146,0.0001934187,0.000041772637,0.00008720125,0.000067954716,0.0015716801,0.0043210364],"genre_scores_gemma":[0.5754963,0.00019612238,0.41618875,0.00033452947,0.00004406108,0.0006639454,0.0003998398,0.00039801994,0.006278452],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99963784,0.00009268567,0.000013781829,0.00007809725,0.00012556369,0.000051920593],"domain_scores_gemma":[0.9991549,0.00045408696,0.000075487726,0.00010559902,0.00012451924,0.000085387604],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008842887,0.0007863097,0.0006509043,0.00040736998,0.00043670204,0.00047541293,0.0018857097,0.0011652396,0.007442932],"category_scores_gemma":[0.0041978247,0.0004118644,0.00040466836,0.0002622143,0.0009261395,0.0011397213,0.0019875462,0.002101425,0.0010006622],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009587919,0.00014326099,0.00044262037,0.00012906632,0.000013942885,0.000055698933,0.00008529331,0.8600625,0.0031263214,0.04798127,0.0030436292,0.08482053],"study_design_scores_gemma":[0.000010497769,0.000025251933,0.000026879747,0.0000065663426,0.0000014881989,0.0000065340087,0.0000025907564,0.99402326,0.00050154875,0.0046460303,0.00074616994,0.0000032546136],"about_ca_topic_score_codex":0.0039429734,"about_ca_topic_score_gemma":0.0046073827,"teacher_disagreement_score":0.007442932,"about_ca_system_score_codex":0.0010564986,"about_ca_system_score_gemma":0.0019326443,"threshold_uncertainty_score":0.024899065},"labels":[],"label_agreement":null},{"id":"W4306167947","doi":"10.1109/tpami.2022.3213503","title":"Robust Losses for Learning Value Functions","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Pattern Analysis and Machine Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; University of Alberta","funders":"","keywords":"Reinforcement learning; Mean squared error; Outlier; Clipping (morphology); Computer science; Bellman equation; Variance (accounting); Mathematical optimization; Robust statistics; Function (biology); Robust control; Sensitivity (control systems); Robust regression; Robustness (evolution); Mathematics; Artificial intelligence; Statistics; Control system","score_opus":0.03400760162769441,"score_gpt":0.26375318575433804,"score_spread":0.22974558412664364,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4306167947","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0020160307,0.00031371272,0.99583155,0.0002188988,0.000029802633,0.000033193824,0.000036690526,0.00013578289,0.0013843627],"genre_scores_gemma":[0.48541108,0.0022546856,0.49880773,0.0007061682,0.00033505657,0.0008342696,0.00045395625,0.0007290782,0.010468114],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9966718,0.0014122982,0.00020954195,0.0005752255,0.0009104792,0.00022059532],"domain_scores_gemma":[0.9920185,0.005837456,0.0006305622,0.00062436436,0.0007171315,0.00017199061],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006731005,0.0020796321,0.0015623292,0.0012103486,0.0005070758,0.0025820376,0.0018674894,0.0023398665,0.0044578547],"category_scores_gemma":[0.0285521,0.00073663914,0.0010075633,0.0009044162,0.0028445907,0.004106416,0.0027495802,0.0040140487,0.0011113627],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010081627,0.000051925574,0.0003181707,0.00019604208,0.00006339645,0.00007245945,0.000084592175,0.63959527,0.001559252,0.31336954,0.0026521685,0.04193636],"study_design_scores_gemma":[0.000015178335,0.00004642759,0.000061510495,0.00004036805,0.000008435812,0.000022315264,0.000009255953,0.8509784,0.0007116797,0.1467491,0.0013451648,0.000012196705],"about_ca_topic_score_codex":0.0012237844,"about_ca_topic_score_gemma":0.0008304897,"teacher_disagreement_score":0.006731005,"about_ca_system_score_codex":0.0024117152,"about_ca_system_score_gemma":0.0015932415,"threshold_uncertainty_score":0.035597384},"labels":[],"label_agreement":null},{"id":"W4306680367","doi":"10.1609/aiide.v18i1.21966","title":"World Models with an Entity-Based Representation","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Alberta Machine Intelligence Institute; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Computer science; Reinforcement learning; Representation (politics); Architecture; Artificial intelligence; Ideal (ethics); Machine learning","score_opus":0.0656754816214134,"score_gpt":0.2890941038071535,"score_spread":0.2234186221857401,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4306680367","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0074489377,0.00020997383,0.9827313,0.0004949193,0.000066801644,0.00007598521,0.0010917485,0.0018332392,0.006047137],"genre_scores_gemma":[0.4055499,0.00094490376,0.57451385,0.00035476903,0.00008817705,0.00049167813,0.005671659,0.00032124933,0.012063775],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995421,0.000121421996,0.000035201876,0.0001416539,0.00011973953,0.00003989303],"domain_scores_gemma":[0.9993849,0.00014317271,0.00008500154,0.0002448618,0.00010042496,0.000041720425],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00039689534,0.0006615127,0.0006104242,0.0007762997,0.00039894032,0.0022925779,0.0020867863,0.0013551557,0.007359842],"category_scores_gemma":[0.0023777075,0.00049278414,0.0011323177,0.0013560303,0.00061938603,0.0040873806,0.0021701688,0.002099615,0.0024935117],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010124688,0.00008974546,0.0009490081,0.000098088036,0.00006671529,0.00020754964,0.00014560153,0.7241588,0.0016877482,0.17840856,0.008902074,0.08518487],"study_design_scores_gemma":[0.000012626857,0.000019218123,0.00013563878,0.000014965491,0.000017193337,0.000035559093,0.000018159608,0.9572405,0.0005041647,0.031860355,0.010130511,0.000011036576],"about_ca_topic_score_codex":0.0074854996,"about_ca_topic_score_gemma":0.008880466,"teacher_disagreement_score":0.0074854996,"about_ca_system_score_codex":0.00080444844,"about_ca_system_score_gemma":0.0009549178,"threshold_uncertainty_score":0.02462107},"labels":[],"label_agreement":null},{"id":"W4307490062","doi":"10.1613/jair.1.13854","title":"Low-Rank Representation of Reinforcement Learning Policies","year":2022,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Canadian Institute for Advanced Research; McGill University","funders":"Canadian Institute for Advanced Research","keywords":"Reproducing kernel Hilbert space; Reinforcement learning; Representation (politics); Computer science; Embedding; Rank (graph theory); Stability (learning theory); Kernel (algebra); Convergence (economics); Space (punctuation); Hilbert space; Hilbert curve; Mathematical optimization; Artificial intelligence; Theoretical computer science; Machine learning; Mathematics; Algorithm; Discrete mathematics; Economics; Pure mathematics","score_opus":0.15821192415856822,"score_gpt":0.4211595401058433,"score_spread":0.26294761594727506,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4307490062","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0045972276,0.00007607558,0.99440765,0.00011103896,0.000014325825,0.000023247321,0.000044965782,0.00018951407,0.0005359949],"genre_scores_gemma":[0.6688945,0.00044802148,0.32402563,0.00021661766,0.00010388086,0.00042651474,0.0004526389,0.0001605632,0.0052716006],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99887115,0.00047521765,0.00007158433,0.00019404027,0.00027357857,0.000114346614],"domain_scores_gemma":[0.99759007,0.0012047291,0.00037398507,0.00035474022,0.0003543945,0.00012211471],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018072934,0.0009563294,0.0012795092,0.000571706,0.00026487853,0.0013941176,0.0011732717,0.001272023,0.0035402353],"category_scores_gemma":[0.0073978268,0.0003510018,0.0006306767,0.00052483194,0.0011565568,0.0018235822,0.0012145616,0.0020349757,0.0007472243],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007883414,0.000056396348,0.00023693829,0.000096255535,0.00002245873,0.00005208532,0.000049393784,0.8987291,0.0025123435,0.06353839,0.00073504564,0.03389273],"study_design_scores_gemma":[0.0000044398707,0.000022217313,0.00002580938,0.0000032633814,0.0000015790648,0.0000054098987,0.0000021661656,0.9864956,0.0003082547,0.012949492,0.00017787747,0.0000038743046],"about_ca_topic_score_codex":0.0019442267,"about_ca_topic_score_gemma":0.0013384726,"teacher_disagreement_score":0.0035402353,"about_ca_system_score_codex":0.00095477625,"about_ca_system_score_gemma":0.0013255046,"threshold_uncertainty_score":0.011843324},"labels":[],"label_agreement":null},{"id":"W4308364487","doi":"10.1109/tnnls.2022.3217189","title":"Monotonic Quantile Network for Worst-Case Offline Reinforcement Learning","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Neural Networks and Learning Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Toronto","funders":"","keywords":"Quantile; Reinforcement learning; Quantile regression; CVAR; Computer science; Quantile function; Monotonic function; Function (biology); Bellman equation; Q-learning; Mathematical optimization; Offline learning; Artificial intelligence; Expected shortfall; Machine learning; Econometrics; Online learning; Mathematics; Cumulative distribution function; Risk management; Statistics; Economics; Probability density function","score_opus":0.02027412986570837,"score_gpt":0.2429601647000125,"score_spread":0.22268603483430413,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4308364487","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015533849,0.0003144636,0.9813447,0.0002579725,0.000036408328,0.000048797847,0.000051915125,0.00048640973,0.0019254477],"genre_scores_gemma":[0.9159251,0.00028334133,0.07920324,0.00034965234,0.000049939266,0.00024400126,0.0001679275,0.00012787087,0.0036489796],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999067,0.0002990481,0.00004349989,0.0002638023,0.00019092826,0.0001358377],"domain_scores_gemma":[0.9976019,0.0015613415,0.00022242198,0.00016637203,0.0003372266,0.00011068761],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002260657,0.0011139717,0.001229191,0.00041366238,0.0003526922,0.0008757563,0.0017623302,0.0011986404,0.003655047],"category_scores_gemma":[0.007893527,0.00047091214,0.00041147665,0.00039005204,0.0010848328,0.0014373137,0.001226799,0.002315091,0.0004889794],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011599636,0.00006603905,0.0009065384,0.00006631655,0.00003262962,0.00006241293,0.00005274611,0.93898845,0.0012462712,0.012912641,0.0009889768,0.04456093],"study_design_scores_gemma":[0.0000045718525,0.000015535856,0.00004670014,0.0000036007157,0.0000033000867,0.000006474357,0.0000023968125,0.9959848,0.00018952703,0.0036217016,0.00011879724,0.0000025467166],"about_ca_topic_score_codex":0.003931135,"about_ca_topic_score_gemma":0.0033122825,"teacher_disagreement_score":0.003931135,"about_ca_system_score_codex":0.0012834772,"about_ca_system_score_gemma":0.0013652672,"threshold_uncertainty_score":0.012227356},"labels":[],"label_agreement":null},{"id":"W4309997788","doi":"10.23919/wmnc56391.2022.9954301","title":"Multi-Agent Actor-Critic for Cooperative Resource Allocation in Vehicular Networks","year":2022,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Computer science; Reinforcement learning; Resource allocation; Quality of service; Scarcity; Distributed computing; Resource (disambiguation); Resource management (computing); Computer network; Artificial intelligence","score_opus":0.02763291788261548,"score_gpt":0.2692623007355205,"score_spread":0.241629382852905,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4309997788","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034445558,0.0006368045,0.9599745,0.00036027047,0.00008034513,0.00004361878,0.000029653203,0.00027752676,0.00415165],"genre_scores_gemma":[0.9683529,0.00017690878,0.028532721,0.000093847724,0.000030099669,0.00009758573,0.00004094312,0.0000296984,0.0026452774],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99948907,0.00021317283,0.000023694003,0.0000980256,0.00009361915,0.00008229117],"domain_scores_gemma":[0.9983827,0.0011068153,0.00016980889,0.00005258518,0.00019964727,0.000088491266],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014730288,0.0011228652,0.0010995818,0.0003509846,0.00039286318,0.0007508093,0.0011482976,0.0010579333,0.0011248839],"category_scores_gemma":[0.0029066398,0.00044687197,0.00045148254,0.00032606223,0.0010401817,0.0005051153,0.00096333167,0.0011876051,0.00017667034],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000020167194,0.000008299664,0.00015394783,0.000016616443,0.00001365505,0.00003386032,0.000014382087,0.9944817,0.0002840459,0.0017904287,0.00014276379,0.0030402748],"study_design_scores_gemma":[0.0000030215772,0.0000057849174,0.000018112252,0.0000013898373,0.0000019025922,0.0000023538712,0.0000017396145,0.9992798,0.000047455975,0.00058193685,0.000055521887,0.0000011627884],"about_ca_topic_score_codex":0.007843683,"about_ca_topic_score_gemma":0.005752818,"teacher_disagreement_score":0.007843683,"about_ca_system_score_codex":0.000939977,"about_ca_system_score_gemma":0.0011629774,"threshold_uncertainty_score":0.015596032},"labels":[],"label_agreement":null},{"id":"W4310608701","doi":"10.1002/9781119808602.index","title":"Index","year":2022,"lang":"en","type":"paratext","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Index (typography); Library science; Computer science; World Wide Web","score_opus":0.017327383746605786,"score_gpt":0.2601855583018081,"score_spread":0.24285817455520234,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4310608701","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00068980874,0.0015513462,0.0096008405,0.002393766,0.0046185143,0.00022475388,0.010440551,0.0020670684,0.9684133],"genre_scores_gemma":[0.0033860812,0.001147609,0.0023150868,0.0005140331,0.00092242175,0.00012447078,0.00540387,0.00064142834,0.9855451],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99939084,0.00008909319,0.000029000335,0.00011361012,0.0003207384,0.00005682649],"domain_scores_gemma":[0.99822253,0.00045841062,0.00007227186,0.00025562168,0.00063822896,0.00035296252],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0005450025,0.0011479827,0.0010516093,0.002673942,0.0013007028,0.0046695773,0.0014119355,0.0016548368,0.7834737],"category_scores_gemma":[0.0049240524,0.0002940418,0.000421873,0.0036338642,0.00049093855,0.0033037064,0.0017214918,0.0011709449,0.6810055],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000030947936,0.000046210593,0.000106807085,0.00017600227,0.0000036413883,0.000018980241,0.000018382409,0.00033801462,0.00028047996,0.011115977,0.86643064,0.12143393],"study_design_scores_gemma":[0.000010322518,0.000015695427,0.00017580483,0.00010665411,0.0000037570396,0.00002831753,0.000022923316,0.0007007317,0.0002281008,0.010071765,0.98862994,0.0000059171502],"about_ca_topic_score_codex":0.00305016,"about_ca_topic_score_gemma":0.0047394875,"teacher_disagreement_score":0.21652633,"about_ca_system_score_codex":0.001536313,"about_ca_system_score_gemma":0.0011968081,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W4310903337","doi":"10.18280/jesa.550514","title":"Simulation of Reinforcement Learning Algorithm for Motion Control of an Autonomous Humanoid","year":2022,"lang":"en","type":"article","venue":"Journal Européen des Systèmes Automatisés","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Reinforcement learning; Markov decision process; Q-learning; Computer science; Humanoid robot; Artificial intelligence; Task (project management); Robot; Robotics; Autonomy; Controller (irrigation); Action (physics); Process (computing); Action selection; Markov chain; Machine learning; Human–computer interaction; Markov process; Engineering; Mathematics","score_opus":0.02035538598490284,"score_gpt":0.26879168026521866,"score_spread":0.24843629428031583,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4310903337","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24292627,0.00078901905,0.7308134,0.0006170293,0.00021466792,0.00021740724,0.00024953522,0.0015124573,0.022660242],"genre_scores_gemma":[0.9745072,0.00011247034,0.022072695,0.00003413782,0.000007793992,0.00015746363,0.00009065715,0.0000187236,0.002998863],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998385,0.000051145504,0.000009742578,0.000028843879,0.0000360813,0.000035567675],"domain_scores_gemma":[0.9995134,0.00027340042,0.000050865805,0.000021848151,0.00010103061,0.000039417275],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00038095785,0.0005631249,0.00055738894,0.00030992128,0.00046694954,0.0005308515,0.00065744587,0.00085624284,0.0053149173],"category_scores_gemma":[0.0012205156,0.00021336287,0.00047152842,0.0001633265,0.00044956652,0.0003353316,0.00059952616,0.00067374506,0.00029234312],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000059238384,0.000019939365,0.0005478624,0.000032787444,0.00001017869,0.00007721774,0.000034550383,0.9931479,0.00069551927,0.0016206191,0.00018903028,0.0035651666],"study_design_scores_gemma":[0.000008641387,0.000020277625,0.00005668071,0.0000027931728,0.0000020308266,0.0000042678544,0.000004423158,0.999298,0.00014570248,0.00030820398,0.00014748395,0.0000015322499],"about_ca_topic_score_codex":0.01263946,"about_ca_topic_score_gemma":0.0052066133,"teacher_disagreement_score":0.01263946,"about_ca_system_score_codex":0.00051463983,"about_ca_system_score_gemma":0.00068940193,"threshold_uncertainty_score":0.025131762},"labels":[],"label_agreement":null},{"id":"W4311681064","doi":"10.22215/etd/2022-15178","title":"Learning Transition Dynamics via Rewarded Exploration: A Study using Unity's MLAgents","year":2022,"lang":"en","type":"dissertation","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Popularity; Artificial intelligence; Hyperparameter; Transition (genetics); Action (physics); Dynamics (music); Variable (mathematics); Focus (optics); Artificial neural network; State (computer science); Game engine; Machine learning; Work (physics); Human–computer interaction; Engineering; Psychology; Social psychology","score_opus":0.03340026244345544,"score_gpt":0.30001871666234387,"score_spread":0.2666184542188884,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4311681064","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9251538,0.0011690896,0.06476918,0.0005331555,0.00004295181,0.0001804504,0.00013692718,0.00034100728,0.007673535],"genre_scores_gemma":[0.97202456,0.00025676462,0.025146563,0.00009965834,0.000014178948,0.000075601485,0.00013484954,0.00005112165,0.002196592],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99865013,0.0007745144,0.000073667296,0.00024017629,0.00017942181,0.00008211959],"domain_scores_gemma":[0.9833824,0.014120033,0.00052028947,0.001137857,0.00054162677,0.00029784883],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027044965,0.00046704945,0.0005890937,0.0004538495,0.00045873338,0.0014484263,0.0015326374,0.0011121528,0.0024112402],"category_scores_gemma":[0.027049826,0.0003759092,0.0005808273,0.0005289012,0.0011798734,0.003079236,0.001463542,0.001854247,0.0004081643],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002168506,0.0051819067,0.040801227,0.0015342381,0.0004933859,0.000631091,0.0070068417,0.55963457,0.010312977,0.055106115,0.005588894,0.3115402],"study_design_scores_gemma":[0.00017820169,0.0013318488,0.0054079345,0.000063661864,0.00006107781,0.00012546896,0.00065790483,0.972047,0.0043932726,0.011577987,0.0041137724,0.00004183162],"about_ca_topic_score_codex":0.007968894,"about_ca_topic_score_gemma":0.0043808953,"teacher_disagreement_score":0.007968894,"about_ca_system_score_codex":0.0008880605,"about_ca_system_score_gemma":0.0005699253,"threshold_uncertainty_score":0.01584506},"labels":[],"label_agreement":null},{"id":"W4312372616","doi":"10.1109/aeeca55500.2022.9918998","title":"Exploration Methods in Reinforcement Learning","year":2022,"lang":"en","type":"article","venue":"2022 IEEE International Conference on Advances in Electrical Engineering and Computer Applications (AEECA)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Reinforcement; Strengths and weaknesses; Error-driven learning; Focus (optics); Artificial intelligence; Human–computer interaction; Engineering; Psychology","score_opus":0.026318313508834495,"score_gpt":0.326034692121068,"score_spread":0.2997163786122335,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312372616","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002021044,0.004838028,0.9833847,0.0006671561,0.00019227898,0.00006211155,0.000030093015,0.00014534644,0.00865915],"genre_scores_gemma":[0.5322445,0.0128543405,0.43391857,0.0009337204,0.0011295107,0.0010881592,0.0001580178,0.00020813027,0.017464941],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99832743,0.00085595244,0.00008921891,0.00019478935,0.0004460027,0.00008648474],"domain_scores_gemma":[0.9970747,0.002184784,0.00021192904,0.00014422125,0.0002838113,0.00010056195],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027554925,0.0012909054,0.0012516868,0.0006355734,0.00041600663,0.0014294138,0.0012484114,0.001365303,0.0037358317],"category_scores_gemma":[0.0066181696,0.00040859196,0.00074620993,0.000792913,0.00221286,0.0019110888,0.0015120936,0.0027088043,0.00074919045],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000080293474,0.000084321,0.0005894124,0.00048326145,0.00011430346,0.000082401806,0.00018183321,0.33257136,0.00094769394,0.5326953,0.0041206563,0.1280491],"study_design_scores_gemma":[0.00007246387,0.000111945395,0.00013457489,0.00010853369,0.00002649171,0.000058477603,0.000027050848,0.6398276,0.000626015,0.34571767,0.013264303,0.000024759418],"about_ca_topic_score_codex":0.0018900901,"about_ca_topic_score_gemma":0.0011252895,"teacher_disagreement_score":0.0037358317,"about_ca_system_score_codex":0.0013293694,"about_ca_system_score_gemma":0.0011761267,"threshold_uncertainty_score":0.01457262},"labels":[],"label_agreement":null},{"id":"W4312374764","doi":"10.1007/978-3-031-18192-4_2","title":"Investigating Effects of Centralized Learning Decentralized Execution on Team Coordination in the Level Based Foraging Environment as a Sequential Social Dilemma","year":2022,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Foraging; Computer science; Dilemma; Social dilemma; Reinforcement learning; Artificial intelligence; Convergence (economics); Social learning; Action (physics); State (computer science); Machine learning; Knowledge management; Ecology; Algorithm; Psychology; Social psychology","score_opus":0.024871536777855238,"score_gpt":0.2559807411380916,"score_spread":0.23110920436023635,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312374764","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9905748,0.000087748245,0.00490421,0.00017735203,0.000015463233,0.00002358365,0.000017841201,0.00001901475,0.004180019],"genre_scores_gemma":[0.99727184,0.00003599276,0.001997627,0.000014226923,0.000005223949,0.000015124938,0.000009966892,0.0000073716296,0.0006426116],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999363,0.0003543782,0.000020469475,0.00008929162,0.00007986085,0.00009304138],"domain_scores_gemma":[0.9829305,0.014956029,0.00086349633,0.00036746956,0.0003046668,0.00057786383],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001420608,0.00037629338,0.0004376885,0.00022373315,0.0003344112,0.0009370008,0.0007863475,0.0006107672,0.0030557374],"category_scores_gemma":[0.012818172,0.00021586666,0.00022278969,0.00024536665,0.00092655345,0.0012775796,0.0007731557,0.0011003336,0.00011617319],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00616074,0.0032537999,0.012141975,0.0003859635,0.00020365904,0.0004504419,0.00079868734,0.86631334,0.027607083,0.039064668,0.0016352179,0.0419845],"study_design_scores_gemma":[0.0002950781,0.0019919833,0.0067682043,0.00001790906,0.000059267513,0.00003982244,0.00052996393,0.9654018,0.0028513866,0.021782085,0.00023668172,0.000025876629],"about_ca_topic_score_codex":0.0025915166,"about_ca_topic_score_gemma":0.0020267179,"teacher_disagreement_score":0.0030557374,"about_ca_system_score_codex":0.0008588826,"about_ca_system_score_gemma":0.0008057296,"threshold_uncertainty_score":0.010222495},"labels":[],"label_agreement":null},{"id":"W4312683699","doi":"10.1109/case49997.2022.9926520","title":"Pareto Frontier Approximation Network (PA-Net) to Solve Bi-objective TSP","year":2022,"lang":"en","type":"article","venue":"2022 IEEE 18th International Conference on Automation Science and Engineering (CASE)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Pareto principle; Mathematical optimization; Computer science; Multi-objective optimization; Reinforcement learning; Set (abstract data type); Metric (unit); Scheduling (production processes); Optimization problem; Job shop scheduling; Mathematics; Artificial intelligence; Schedule; Engineering","score_opus":0.025091202479126374,"score_gpt":0.26447059761505104,"score_spread":0.23937939513592466,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312683699","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03647857,0.0006690573,0.95169944,0.00047753533,0.000118035176,0.000113139795,0.00023985388,0.0013191852,0.008885279],"genre_scores_gemma":[0.6144699,0.00056066574,0.37175918,0.000622108,0.00006854997,0.00052480574,0.0012138995,0.00029668355,0.010484125],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99968207,0.000100932855,0.000016230686,0.00007705327,0.000068192145,0.000055619934],"domain_scores_gemma":[0.99924856,0.00044689898,0.000057713874,0.000044830016,0.00015521966,0.000046743517],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008622516,0.0012864573,0.0010403324,0.0008209003,0.00062827,0.0006912029,0.0012235688,0.0013572294,0.0052449135],"category_scores_gemma":[0.0024450656,0.0005289002,0.0007760999,0.0007157053,0.000665826,0.0010109032,0.00092926546,0.0015154089,0.0007343625],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000047003454,0.000040246916,0.00035491085,0.00004341085,0.000025984966,0.000046280205,0.000027878066,0.9615243,0.00042146584,0.0038659656,0.0019536421,0.031648934],"study_design_scores_gemma":[0.0000040448567,0.000011108426,0.000025207992,0.0000039163215,0.000003168239,0.0000055671503,0.0000033097679,0.99787545,0.000108677275,0.0016698234,0.00028831715,0.0000014323896],"about_ca_topic_score_codex":0.011880564,"about_ca_topic_score_gemma":0.010330414,"teacher_disagreement_score":0.011880564,"about_ca_system_score_codex":0.0012672795,"about_ca_system_score_gemma":0.0016728705,"threshold_uncertainty_score":0.02362287},"labels":[],"label_agreement":null},{"id":"W4312995559","doi":"10.1109/tnsm.2022.3210827","title":"FLoadNet: Load Balancing in Fog Networks With Cooperative Multiagent Using Actor–Critic Method","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Network and Service Management","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Reinforcement learning; Distributed computing; Load balancing (electrical power); Cloud computing; Edge computing; Edge device; Context (archaeology); Enhanced Data Rates for GSM Evolution; Workload; Server; Shared resource; Computer network; Artificial intelligence","score_opus":0.01611502068863893,"score_gpt":0.24853917710541473,"score_spread":0.2324241564167758,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312995559","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027542062,0.0003392875,0.9673431,0.0002598675,0.00012064454,0.00006617322,0.00002653838,0.00049095775,0.0038113978],"genre_scores_gemma":[0.9332878,0.00018275464,0.0626336,0.00019242341,0.000051780276,0.00012547312,0.00006125613,0.000061611165,0.003403362],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99976236,0.000065222746,0.000009682862,0.000058663674,0.000056952944,0.00004716907],"domain_scores_gemma":[0.9994646,0.00027200766,0.000070003276,0.000025658528,0.000112877824,0.00005491445],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00091442943,0.00091998273,0.0009142856,0.00034372328,0.00046337416,0.0008681758,0.0013994437,0.0010534453,0.0012750342],"category_scores_gemma":[0.0014558932,0.0004018582,0.0004486974,0.0002382258,0.0007674379,0.0006976258,0.000931245,0.0010612857,0.00021069554],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000041183674,0.000031768297,0.00040142637,0.000025224557,0.00002540836,0.000060104227,0.000030224091,0.9817931,0.000782658,0.0027690562,0.0006340125,0.013405889],"study_design_scores_gemma":[0.0000041182416,0.0000069595117,0.000017221671,0.0000013145342,0.0000017266789,0.0000026950172,0.0000017801784,0.9993549,0.00007367865,0.00042058516,0.00011387906,0.0000012367964],"about_ca_topic_score_codex":0.010243389,"about_ca_topic_score_gemma":0.0084062535,"teacher_disagreement_score":0.010243389,"about_ca_system_score_codex":0.0008038538,"about_ca_system_score_gemma":0.0010981163,"threshold_uncertainty_score":0.020367503},"labels":[],"label_agreement":null},{"id":"W4313061161","doi":"10.1007/978-3-031-19679-9_63","title":"Reinforcement Learning for Exploring Pedagogical Strategies in Virtual Reality Training","year":2022,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"TUTOR; Reinforcement learning; Computer science; Virtual reality; Reinforcement; Human–computer interaction; Training (meteorology); Multimedia; Artificial intelligence; Psychology","score_opus":0.3056666438607475,"score_gpt":0.3725653755714306,"score_spread":0.06689873171068311,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4313061161","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014255667,0.005768724,0.91770726,0.0004899161,0.00018037108,0.000094268464,0.000056325287,0.00062513945,0.060822286],"genre_scores_gemma":[0.4677099,0.004543985,0.47664854,0.00016947233,0.00010288372,0.0004278225,0.00011247005,0.00015015031,0.050134823],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99978334,0.00011302168,0.0000068434874,0.000028069402,0.00005603779,0.000012774409],"domain_scores_gemma":[0.99959797,0.00032239914,0.000016858208,0.00002041375,0.00002788638,0.000014499806],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00040355313,0.0005596053,0.00032078387,0.00019008173,0.00019795983,0.0008651081,0.00080141664,0.000614049,0.008315252],"category_scores_gemma":[0.0016162721,0.0001439518,0.00023565965,0.00025810336,0.00057952607,0.0008213506,0.0006137219,0.0008746343,0.0008295959],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013886947,0.0001259782,0.00031096875,0.00040632382,0.000037860955,0.00010915843,0.0006029034,0.12130075,0.008249153,0.22170311,0.006845842,0.6401691],"study_design_scores_gemma":[0.00008166982,0.0003204906,0.0010184986,0.00038761247,0.000043561016,0.0003320406,0.000423274,0.64783293,0.008102952,0.2708279,0.070568785,0.000060247527],"about_ca_topic_score_codex":0.001202583,"about_ca_topic_score_gemma":0.0017238403,"teacher_disagreement_score":0.008315252,"about_ca_system_score_codex":0.0006172662,"about_ca_system_score_gemma":0.00038473782,"threshold_uncertainty_score":0.027817309},"labels":[],"label_agreement":null},{"id":"W4313483332","doi":"10.48550/arxiv.2301.00051","title":"Learning from Guided Play: Improving Exploration for Adversarial Imitation Learning with Simple Auxiliary Tasks","year":2022,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Rehabilitation Institute; University of Alberta","funders":"","keywords":"Leverage (statistics); Computer science; Artificial intelligence; Machine learning; Reinforcement learning; Task (project management); Adversarial system; Imitation; Engineering","score_opus":0.09622267453762151,"score_gpt":0.2116297504875747,"score_spread":0.1154070759499532,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4313483332","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039511375,0.000533326,0.95504725,0.00032641727,0.000051195413,0.00009441185,0.00006856029,0.0013760373,0.0029914011],"genre_scores_gemma":[0.90460527,0.00024950807,0.090815954,0.0003034071,0.00005670681,0.00021939943,0.00018041453,0.00020604204,0.0033632105],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99923086,0.00030372522,0.000035450048,0.00016745586,0.00015829383,0.00010413985],"domain_scores_gemma":[0.9966254,0.0023187462,0.00026898648,0.00041288565,0.00018431574,0.00018976581],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019532524,0.0015389653,0.0012395268,0.00041663606,0.0003814795,0.000771966,0.0020945668,0.0013980696,0.002584432],"category_scores_gemma":[0.008356214,0.00065838586,0.0006803624,0.00029451257,0.0018997686,0.0018549081,0.003002337,0.0025187514,0.0006431442],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019805782,0.00014773717,0.0014901592,0.0001255029,0.0000620369,0.00012826682,0.00013098826,0.92930424,0.0031655587,0.01229671,0.0013975494,0.051553186],"study_design_scores_gemma":[0.000011650272,0.00005292766,0.00006291667,0.000008006306,0.000004259487,0.00001506371,0.0000054588572,0.99404013,0.00038861082,0.0051893876,0.00021736855,0.0000041848443],"about_ca_topic_score_codex":0.0025303038,"about_ca_topic_score_gemma":0.0025666566,"teacher_disagreement_score":0.002584432,"about_ca_system_score_codex":0.00077553175,"about_ca_system_score_gemma":0.0011852008,"threshold_uncertainty_score":0.010329962},"labels":[],"label_agreement":null},{"id":"W4315588467","doi":"10.48550/arxiv.2301.02952","title":"Learning Symbolic Representations for Reinforcement Learning of Non-Markovian Behavior","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Government of Canada; Canadian Institute for Advanced Research; Agencia Nacional de Investigación y Desarrollo; Microsoft Research","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Abstraction; Action (physics); Representation (politics); Function (biology); Markov process; State (computer science); Temporal difference learning; Markov decision process; Machine learning; Theoretical computer science; Algorithm; Mathematics","score_opus":0.08623908006050819,"score_gpt":0.24314598183921113,"score_spread":0.15690690177870292,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4315588467","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02832949,0.00024356425,0.9687007,0.0002495848,0.000024324752,0.000038682057,0.00014900872,0.00088528194,0.0013794336],"genre_scores_gemma":[0.8188559,0.00029767994,0.17774622,0.00011942241,0.000030518142,0.00021320312,0.0005414082,0.0001345543,0.002061145],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996321,0.0001236183,0.000023377594,0.0000979886,0.00007860153,0.000044234115],"domain_scores_gemma":[0.99797946,0.0014278184,0.00018665558,0.00021822873,0.00011804924,0.00006988272],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00063179777,0.00066021207,0.0007134179,0.00053310284,0.00028887787,0.00078777026,0.0011240311,0.0007444279,0.0030455638],"category_scores_gemma":[0.005093711,0.00041141643,0.00061235996,0.0004993194,0.0010574054,0.0016926224,0.001089553,0.001893525,0.00044549225],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008618954,0.00006654269,0.0010143389,0.00009076063,0.000033939712,0.00005900305,0.000110748115,0.8939163,0.001356845,0.03762094,0.00083957677,0.064804845],"study_design_scores_gemma":[0.0000056567715,0.000010339301,0.000037282498,0.0000063106568,0.0000029433666,0.000004237967,0.000005929156,0.97844213,0.00026828222,0.021014938,0.00019966402,0.0000022877666],"about_ca_topic_score_codex":0.0035551991,"about_ca_topic_score_gemma":0.0063289367,"teacher_disagreement_score":0.0035551991,"about_ca_system_score_codex":0.0012537194,"about_ca_system_score_gemma":0.0010555419,"threshold_uncertainty_score":0.010188401},"labels":[],"label_agreement":null},{"id":"W4315798474","doi":"10.1007/978-3-031-19907-3_31","title":"Maze Learning Using a Hyperdimensional Predictive Processing Cognitive Architecture","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Cognitive architecture; Architecture; Cognition; Artificial intelligence; Cognitive science; Human–computer interaction; Psychology; Neuroscience","score_opus":0.026256872872513812,"score_gpt":0.2637810035091903,"score_spread":0.2375241306366765,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4315798474","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06263138,0.00027303613,0.9253496,0.00029059447,0.000098671975,0.000058214857,0.00008613827,0.00091027335,0.010302121],"genre_scores_gemma":[0.7179766,0.00033866748,0.274409,0.00010231262,0.000044540608,0.00013743319,0.0001199298,0.00006476092,0.0068067145],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99990094,0.0000157549,0.000006229268,0.00003196984,0.00003072194,0.000014400375],"domain_scores_gemma":[0.9996871,0.00012224067,0.000026747917,0.000057324905,0.000076700424,0.000029896453],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00022080501,0.00045451106,0.00046267282,0.00024807182,0.00034493505,0.0011985205,0.0013486217,0.00056469487,0.0039366037],"category_scores_gemma":[0.00083568564,0.00029188162,0.0005842705,0.00044790556,0.00056129385,0.0015052245,0.0009853509,0.0011256043,0.00037487654],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002084992,0.00015005737,0.0008099386,0.00011872101,0.00011252259,0.00014155296,0.00015773627,0.60612315,0.023784846,0.11107926,0.0028949266,0.2544188],"study_design_scores_gemma":[0.000011391525,0.000046077377,0.0001399868,0.000006781244,0.000013667081,0.00001780751,0.000009552313,0.968411,0.0018654302,0.028833972,0.0006332532,0.000010981609],"about_ca_topic_score_codex":0.0031548743,"about_ca_topic_score_gemma":0.00313309,"teacher_disagreement_score":0.0039366037,"about_ca_system_score_codex":0.00053624675,"about_ca_system_score_gemma":0.0005869262,"threshold_uncertainty_score":0.013169289},"labels":[],"label_agreement":null},{"id":"W4316464918","doi":"10.3390/robotics12010012","title":"Simulated and Real Robotic Reach, Grasp, and Pick-and-Place Using Combined Reinforcement Learning and Traditional Controls","year":2023,"lang":"en","type":"article","venue":"Robotics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Korea Electrotechnology Research Institute","keywords":"Reinforcement learning; GRASP; Task (project management); Artificial intelligence; Robotics; Computer science; Robot; Control (management); Plan (archaeology); Margin (machine learning); Human–computer interaction; Machine learning; Engineering; Software engineering; Systems engineering","score_opus":0.03761961539231777,"score_gpt":0.26665880430827926,"score_spread":0.22903918891596148,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4316464918","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.73229665,0.00028108308,0.25163382,0.00030415523,0.00008165182,0.00022078038,0.00021371603,0.0011498934,0.01381834],"genre_scores_gemma":[0.97405416,0.00006149125,0.024324978,0.000017298431,0.000003589694,0.00009621424,0.00006176855,0.000022611468,0.0013578162],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99961853,0.00015271867,0.00001578305,0.000054195207,0.00011527777,0.000043546035],"domain_scores_gemma":[0.9991246,0.0005579472,0.000065143475,0.000113958515,0.000081461505,0.000056848745],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006431739,0.0005136712,0.00033514304,0.00030309244,0.00023363251,0.0004610489,0.00078193995,0.0006041356,0.0015891569],"category_scores_gemma":[0.0015641967,0.00025941932,0.0003901372,0.00021943307,0.0011193426,0.00055503723,0.0005488333,0.0006529581,0.00014891158],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024639376,0.00020684263,0.0006542955,0.00007514685,0.00003180601,0.000071917304,0.00006464375,0.9800587,0.0041728704,0.0028534376,0.00025993458,0.011304038],"study_design_scores_gemma":[0.00007365782,0.00040052296,0.0006965813,0.000009224467,0.000009897778,0.000023907858,0.000034266355,0.98933303,0.0062762643,0.0020531157,0.0010747995,0.000014787346],"about_ca_topic_score_codex":0.0049344627,"about_ca_topic_score_gemma":0.006378041,"teacher_disagreement_score":0.0049344627,"about_ca_system_score_codex":0.00076505565,"about_ca_system_score_gemma":0.0006255708,"threshold_uncertainty_score":0.009811461},"labels":[],"label_agreement":null},{"id":"W4318620677","doi":"10.48550/arxiv.2301.11490","title":"Neural Episodic Control with State Abstraction","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"JST-Mirai Program; Japan Society for the Promotion of Science; Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China; Canadian Institute for Advanced Research","keywords":"Abstraction; Computer science; Leverage (statistics); Episodic memory; Reinforcement learning; Artificial intelligence; Sample (material); State (computer science); Inefficiency; Control (management); Machine learning; Scalability; Programming language; Psychology; Database","score_opus":0.07064646825082561,"score_gpt":0.18665949019521444,"score_spread":0.11601302194438882,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4318620677","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032307565,0.0002734105,0.962805,0.00013321367,0.000053783737,0.0000614137,0.00007616057,0.0013442255,0.0029451754],"genre_scores_gemma":[0.9189457,0.00011784413,0.078161046,0.00012893541,0.000030758576,0.00012229674,0.00015892029,0.00006274995,0.0022717153],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99960417,0.00008004065,0.000028280436,0.0001252644,0.00010348235,0.000058934747],"domain_scores_gemma":[0.9991817,0.00033476192,0.000110683715,0.00018589568,0.0001328159,0.000054179014],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008480549,0.00068418565,0.00068671315,0.0003030569,0.00025132595,0.00068461354,0.0013170118,0.0005269184,0.0023269255],"category_scores_gemma":[0.0025335443,0.00028511102,0.0003996729,0.00030482042,0.000859375,0.0010264482,0.0012262068,0.0011613872,0.0003040852],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014114744,0.0001046808,0.0011432453,0.00009042494,0.000050697487,0.00006124535,0.00007730134,0.8169487,0.0044471994,0.012292955,0.0015665634,0.16307585],"study_design_scores_gemma":[0.000009389027,0.000027972374,0.000089384324,0.0000037718362,0.0000049963946,0.000008499818,0.0000028910986,0.9954052,0.00075634324,0.0033575718,0.00033053046,0.000003488518],"about_ca_topic_score_codex":0.004110607,"about_ca_topic_score_gemma":0.004684545,"teacher_disagreement_score":0.004110607,"about_ca_system_score_codex":0.00067943224,"about_ca_system_score_gemma":0.0009091552,"threshold_uncertainty_score":0.0081733465},"labels":[],"label_agreement":null},{"id":"W4318829810","doi":"10.2139/ssrn.4229967","title":"Combining Information-Seeking Exploration and Reward Maximization: Unified Inference on Continuous State and Action Spaces Under Partial Observability","year":2022,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Observability; Inference; Action (physics); Maximization; State (computer science); Utility maximization; Mathematical economics; Complete information; Computer science; Artificial intelligence; Psychology; Microeconomics; Social psychology; Economics; Mathematics; Algorithm; Applied mathematics","score_opus":0.031766764619397814,"score_gpt":0.2606905869329649,"score_spread":0.22892382231356706,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4318829810","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037384447,0.00039024794,0.960816,0.00035286404,0.000025302164,0.000037068545,0.00009177161,0.00018415874,0.00071817776],"genre_scores_gemma":[0.9105348,0.0003836489,0.086893484,0.00017360903,0.00015154484,0.000173935,0.00030337408,0.00008607337,0.0012995519],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99796706,0.00076884666,0.00015061566,0.0005396293,0.0003287462,0.00024517052],"domain_scores_gemma":[0.97993463,0.01696778,0.001103805,0.0007927079,0.0006419557,0.000559207],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005553326,0.0015980554,0.004222459,0.0017871114,0.0006812708,0.0025337916,0.0034277663,0.0028295184,0.0016602988],"category_scores_gemma":[0.024745215,0.002200088,0.0020724977,0.0017968528,0.0029589057,0.0060802535,0.004333419,0.0033090855,0.00022543398],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002838233,0.00015136259,0.0018165149,0.00015532757,0.00019146637,0.000101556,0.000108561,0.9360973,0.0006294463,0.032803617,0.00046272363,0.02719828],"study_design_scores_gemma":[0.000010791732,0.000017669407,0.00010930741,0.0000058217365,0.000009402713,0.0000049679447,0.000002813812,0.9824829,0.0000911081,0.017232867,0.000025947289,0.000006483211],"about_ca_topic_score_codex":0.0076659652,"about_ca_topic_score_gemma":0.007357811,"teacher_disagreement_score":0.0076659652,"about_ca_system_score_codex":0.0014520115,"about_ca_system_score_gemma":0.002809031,"threshold_uncertainty_score":0.029369116},"labels":[],"label_agreement":null},{"id":"W4319441257","doi":"10.1016/j.neucom.2023.01.076","title":"Uncertainty-aware transfer across tasks using hybrid model-based successor feature reinforcement learning☆","year":2023,"lang":"en","type":"article","venue":"Neurocomputing","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Defence Research and Development Canada; University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Successor cardinal; Generalization; Artificial intelligence; Feature (linguistics); Machine learning; Sample (material); Knowledge transfer; Kalman filter; Stability (learning theory); Mathematics","score_opus":0.033756259061986336,"score_gpt":0.29683880846576377,"score_spread":0.26308254940377745,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319441257","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07660862,0.00018801598,0.917935,0.00019109304,0.00007819522,0.000078688136,0.000040571096,0.0010565899,0.003823195],"genre_scores_gemma":[0.96874475,0.000033583292,0.029698975,0.000039207764,0.0000124895105,0.000062425526,0.00003213525,0.000037410096,0.0013390207],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996177,0.00007783232,0.000019963467,0.00009852598,0.00011165394,0.000074244184],"domain_scores_gemma":[0.9989982,0.00046344483,0.00012059114,0.00015477212,0.00017898454,0.00008395144],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008251238,0.0006448692,0.0010776448,0.00029663678,0.00042098016,0.0007118738,0.0014198174,0.00080090895,0.002306527],"category_scores_gemma":[0.0024743597,0.0004203832,0.0005000152,0.00025501347,0.00064158544,0.00096241565,0.0016687139,0.0011502508,0.0003824499],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024600414,0.00019601268,0.0006026312,0.00006443077,0.000055146793,0.0001000712,0.000095579984,0.8740503,0.0080948975,0.0054887463,0.0010919528,0.10991428],"study_design_scores_gemma":[0.0000060587777,0.000026894779,0.00005698432,0.0000018974658,0.0000032075989,0.0000072421326,0.0000025101933,0.9980872,0.0004695476,0.0012579522,0.000077257,0.0000030568267],"about_ca_topic_score_codex":0.0044639194,"about_ca_topic_score_gemma":0.0038047312,"teacher_disagreement_score":0.0044639194,"about_ca_system_score_codex":0.0006532755,"about_ca_system_score_gemma":0.00103888,"threshold_uncertainty_score":0.008875847},"labels":[],"label_agreement":null},{"id":"W4319985622","doi":"10.1002/cjce.24878","title":"Multi‐agent reinforcement learning for process control: Exploring the intersection between fields of reinforcement learning, control theory, and game theory","year":2023,"lang":"en","type":"article","venue":"The Canadian Journal of Chemical Engineering","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"University of Alberta","keywords":"Reinforcement learning; Marl; Controller (irrigation); Computer science; Process (computing); Function (biology); Control (management); Intersection (aeronautics); Reinforcement; Control system; Control theory (sociology); Control engineering; Artificial intelligence; Engineering","score_opus":0.02187824374282683,"score_gpt":0.2356329742072273,"score_spread":0.2137547304644005,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319985622","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031264987,0.00071659934,0.9641102,0.000592593,0.00004244028,0.000048233436,0.000006501294,0.000104261446,0.0031141876],"genre_scores_gemma":[0.9369969,0.00039664854,0.06134588,0.00013676533,0.00005616597,0.00010119647,0.0000099194995,0.000019890937,0.00093664514],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9994344,0.0002851218,0.00002306008,0.000085962674,0.000119928445,0.00005137998],"domain_scores_gemma":[0.99754566,0.0018291126,0.00021920951,0.00009623503,0.00021566862,0.000094050585],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017376149,0.00059408514,0.0008639523,0.00036492318,0.00032117649,0.0010042824,0.00082617265,0.0008883164,0.0011268953],"category_scores_gemma":[0.0035054553,0.00033967738,0.00050139794,0.00028698347,0.0014340278,0.0010044556,0.0010686148,0.0012818298,0.00011125178],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004177134,0.00006896407,0.00045941575,0.000072927876,0.000035219033,0.000050074435,0.000048197926,0.960845,0.0014727643,0.023540145,0.00016141319,0.013204186],"study_design_scores_gemma":[0.0000051152147,0.000022661046,0.00003113809,0.000004171673,0.0000020321477,0.0000029241055,0.0000031985019,0.99560404,0.0001425809,0.0040554316,0.0001244563,0.0000024000515],"about_ca_topic_score_codex":0.0031068928,"about_ca_topic_score_gemma":0.0016096422,"teacher_disagreement_score":0.0031068928,"about_ca_system_score_codex":0.0008693811,"about_ca_system_score_gemma":0.0010334374,"threshold_uncertainty_score":0.0091894865},"labels":[],"label_agreement":null},{"id":"W4319996478","doi":"10.1109/tg.2023.3237943","title":"Predictive Dead Reckoning for Online Peer-to-Peer Games","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Games","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ubisoft (Canada); University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Dead reckoning; Peer-to-peer; Computer science; Internet privacy; World Wide Web; Telecommunications; Global Positioning System","score_opus":0.03682903211344726,"score_gpt":0.3037323543981614,"score_spread":0.26690332228471414,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319996478","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07509097,0.0002780122,0.91992164,0.00019427156,0.000064375636,0.00007524114,0.000038068294,0.000636569,0.0037008615],"genre_scores_gemma":[0.9724998,0.00006997267,0.02558096,0.00003461716,0.000012105994,0.000055553905,0.00002627862,0.000022756654,0.0016980763],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99960214,0.00009189561,0.000022016508,0.000087486675,0.0001397408,0.000056625497],"domain_scores_gemma":[0.9987853,0.0006891709,0.00015022409,0.000110137444,0.00017458154,0.00009058907],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000670747,0.00073397177,0.00080875534,0.0003346357,0.0005846204,0.00062707363,0.0016268332,0.0008056305,0.0013652275],"category_scores_gemma":[0.0031966055,0.00034514477,0.0002459698,0.0002284827,0.0010292247,0.0014119113,0.0012826865,0.0011035659,0.00021838193],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011599497,0.000062482286,0.00044383228,0.000042509088,0.00001887076,0.00010477754,0.00011830037,0.96041703,0.002131638,0.006388333,0.00042193488,0.029734237],"study_design_scores_gemma":[0.000007002365,0.000022810478,0.00005222795,0.000002015464,0.0000020377677,0.000009461982,0.000009789645,0.9964103,0.00035635466,0.0029702918,0.00015443804,0.0000033485405],"about_ca_topic_score_codex":0.007112288,"about_ca_topic_score_gemma":0.00636005,"teacher_disagreement_score":0.007112288,"about_ca_system_score_codex":0.00067941286,"about_ca_system_score_gemma":0.00070460234,"threshold_uncertainty_score":0.014141798},"labels":[],"label_agreement":null},{"id":"W4320204347","doi":"10.48550/arxiv.2206.05860","title":"IGN : Implicit Generative Networks","year":2022,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Science North","funders":"","keywords":"Discriminator; Reinforcement learning; Quantile regression; Computer science; Generator (circuit theory); Quantile; Baseline (sea); Bellman equation; Artificial intelligence; State (computer science); Function (biology); Generative grammar; Action (physics); Mathematical optimization; Machine learning; Econometrics; Algorithm; Mathematics; Power (physics)","score_opus":0.06671544719590597,"score_gpt":0.19403896368444742,"score_spread":0.12732351648854145,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4320204347","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009300726,0.00048940594,0.97258323,0.00082079746,0.00013558051,0.000050660317,0.0005977853,0.0029043127,0.013117526],"genre_scores_gemma":[0.6933976,0.000849764,0.26810324,0.0010269778,0.00020730718,0.0003484366,0.0026418031,0.0014872904,0.031937573],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995105,0.00017549566,0.000015059841,0.00013859171,0.00010910801,0.000051237872],"domain_scores_gemma":[0.9989353,0.0006720135,0.000051487063,0.00018885595,0.00009137303,0.00006107166],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00082869764,0.00087978016,0.00085497793,0.0005673754,0.00048387397,0.0011663368,0.0019864144,0.0013284917,0.01214507],"category_scores_gemma":[0.0047811996,0.00055883115,0.00078562665,0.0006687326,0.0012377917,0.0020591551,0.0023409585,0.0027366506,0.003057238],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001103766,0.0000841186,0.0012288757,0.00013380659,0.000064463806,0.00020222194,0.00011519611,0.6418924,0.0014141347,0.24018359,0.012393266,0.102177635],"study_design_scores_gemma":[0.000010429438,0.0000093004755,0.00006289924,0.000011468072,0.0000056889603,0.00003358864,0.0000048278357,0.8902267,0.0003118539,0.10603225,0.0032859258,0.0000050817966],"about_ca_topic_score_codex":0.0033245084,"about_ca_topic_score_gemma":0.0068374476,"teacher_disagreement_score":0.01214507,"about_ca_system_score_codex":0.00115428,"about_ca_system_score_gemma":0.0008999749,"threshold_uncertainty_score":0.040629327},"labels":[],"label_agreement":null},{"id":"W4320342078","doi":"10.48550/arxiv.2302.00237","title":"Bridging Physics-Informed Neural Networks with Reinforcement Learning: Hamilton-Jacobi-Bellman Proximal Policy Optimization (HJBPPO)","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"University of Waterloo","keywords":"Hamilton–Jacobi–Bellman equation; Reinforcement learning; Bellman equation; Artificial neural network; Bridging (networking); Computer science; Mathematical optimization; Optimal control; Mathematics; Artificial intelligence","score_opus":0.05811504637595478,"score_gpt":0.20704281154236612,"score_spread":0.14892776516641135,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4320342078","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0045944625,0.00017521347,0.99276143,0.0002266179,0.00004246098,0.000024189525,0.000008494561,0.0001385602,0.0020284946],"genre_scores_gemma":[0.5702636,0.0005241221,0.42328054,0.00047380207,0.00011995689,0.00025556074,0.00006375366,0.00017197311,0.004846611],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99936014,0.0003213409,0.000021280684,0.00008847204,0.0001558896,0.000052804084],"domain_scores_gemma":[0.9987545,0.00083845237,0.000097798984,0.000102729864,0.00013794078,0.000068514404],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001894337,0.00083741266,0.0010909386,0.00037180484,0.0005166814,0.0009616132,0.0014112325,0.0015663272,0.0020773504],"category_scores_gemma":[0.0053857556,0.0006204315,0.0004414595,0.0004944137,0.0017238132,0.0013331001,0.0020701794,0.0022842328,0.00039155624],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000044108536,0.00004332703,0.0002851064,0.000057933645,0.00003221278,0.000051097355,0.00005354094,0.9046214,0.00076184847,0.0632538,0.0009749522,0.029820671],"study_design_scores_gemma":[0.000007396123,0.0000148791805,0.000026514183,0.0000054225306,0.0000025866793,0.000007277812,0.0000028249121,0.98152244,0.00024216577,0.017690314,0.0004738338,0.000004386339],"about_ca_topic_score_codex":0.0044607944,"about_ca_topic_score_gemma":0.0030067936,"teacher_disagreement_score":0.0044607944,"about_ca_system_score_codex":0.0010065784,"about_ca_system_score_gemma":0.0020927098,"threshold_uncertainty_score":0.010018349},"labels":[],"label_agreement":null},{"id":"W4321472417","doi":"10.48550/arxiv.2302.09465","title":"Stochastic Generative Flow Networks","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Samsung; Genentech; Canadian Institute for Advanced Research","keywords":"Computer science; Probabilistic logic; Inference; Variety (cybernetics); Stochastic modelling; Limit (mathematics); Generative grammar; Sample (material); Flow (mathematics); Stochastic dynamics; Mathematical optimization; Artificial intelligence; Mathematics; Statistical physics","score_opus":0.10064847479664266,"score_gpt":0.19619547860151418,"score_spread":0.09554700380487152,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4321472417","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021193333,0.00042501948,0.9714049,0.00042476237,0.000063539665,0.00006239721,0.00029996398,0.00072742335,0.0053986623],"genre_scores_gemma":[0.7366094,0.00080338604,0.2496124,0.0005447053,0.00011671985,0.00033609947,0.0014051048,0.00037037217,0.010201846],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9994553,0.00017166263,0.000025156543,0.00016970834,0.00010573216,0.000072589886],"domain_scores_gemma":[0.9982128,0.0012016182,0.00016770296,0.00013350887,0.00018271084,0.00010180986],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001073834,0.0010106772,0.0008967726,0.00092169177,0.0006675355,0.0011247001,0.0014780795,0.0011978425,0.0045940904],"category_scores_gemma":[0.0049863374,0.0005147888,0.0009800382,0.00074733753,0.0015169942,0.0016507509,0.0014963556,0.0013677496,0.0006184092],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000039475937,0.000023641363,0.00096198573,0.000057089386,0.00003176394,0.00007687729,0.00005365221,0.89824516,0.0006997754,0.06501122,0.0019101909,0.03288923],"study_design_scores_gemma":[0.000006287077,0.000008585446,0.00007909687,0.000008449874,0.000005481532,0.000016748154,0.00000545414,0.9566417,0.00021498482,0.041968893,0.001039047,0.0000053246713],"about_ca_topic_score_codex":0.005725728,"about_ca_topic_score_gemma":0.007526932,"teacher_disagreement_score":0.005725728,"about_ca_system_score_codex":0.0014172886,"about_ca_system_score_gemma":0.0012181671,"threshold_uncertainty_score":0.01536876},"labels":[],"label_agreement":null},{"id":"W4323020953","doi":"10.1109/access.2023.3249572","title":"Bridging the Reality Gap Between Virtual and Physical Environments Through Reinforcement Learning","year":2023,"lang":"en","type":"article","venue":"IEEE Access","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Computer science; Virtual reality; Bridging (networking); Metaverse; Artificial intelligence; Human–computer interaction; Physics engine; Task (project management); Simulation; Engineering","score_opus":0.08104295423193325,"score_gpt":0.3377379428838238,"score_spread":0.25669498865189055,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4323020953","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07440344,0.00011771509,0.92231196,0.00016569282,0.000033432312,0.00008901003,0.000014540967,0.00076481694,0.0020994642],"genre_scores_gemma":[0.8834175,0.000079253834,0.1153586,0.00009294272,0.000011614223,0.00011159084,0.000032077132,0.00005167578,0.0008448999],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99934465,0.000250155,0.000031812797,0.000144714,0.0001541078,0.00007446021],"domain_scores_gemma":[0.99774534,0.0014360372,0.00030482357,0.00021553495,0.00017789987,0.00012027771],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016403003,0.0008286724,0.0006031294,0.00024592315,0.0003153407,0.0008875821,0.0010974382,0.00063077855,0.001035653],"category_scores_gemma":[0.005199367,0.00044275692,0.00037180466,0.00014905691,0.0012750851,0.0011280123,0.0016106355,0.0015613277,0.00021628015],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021105316,0.00020410435,0.0012428237,0.00008237239,0.000037973252,0.00012101897,0.00016248363,0.9269296,0.007993841,0.006042063,0.00040402182,0.056568682],"study_design_scores_gemma":[0.000018775918,0.000113427064,0.00014988644,0.000011342073,0.0000072726534,0.000019867213,0.000015478015,0.99467385,0.0025274293,0.0019971048,0.0004580882,0.000007366091],"about_ca_topic_score_codex":0.0019831124,"about_ca_topic_score_gemma":0.0015633337,"teacher_disagreement_score":0.0019831124,"about_ca_system_score_codex":0.00064398075,"about_ca_system_score_gemma":0.0009294252,"threshold_uncertainty_score":0.00867486},"labels":[],"label_agreement":null},{"id":"W4327810476","doi":"10.48550/arxiv.2303.09032","title":"Conditionally Optimistic Exploration for Cooperative Deep Multi-Agent Reinforcement Learning","year":2023,"lang":"en","type":"preprint","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Institut de Valorisation des Données; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Computer science; Monte Carlo tree search; Intuition; Tree (set theory); Artificial intelligence; Software deployment; Machine learning; Mathematical optimization; Mathematics; Cognitive science","score_opus":0.04615468709767459,"score_gpt":0.28443516501314803,"score_spread":0.23828047791547344,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4327810476","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028045122,0.00016200797,0.97003615,0.00016874849,0.000018947398,0.000036050675,0.00001628543,0.00038568644,0.0011310022],"genre_scores_gemma":[0.8828496,0.00007775189,0.115564965,0.00010223608,0.000021170206,0.0001347803,0.000044292065,0.00006056802,0.0011447027],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992035,0.0003516119,0.00003999484,0.00013190201,0.00017317684,0.00009979757],"domain_scores_gemma":[0.99722666,0.0017533202,0.0003129922,0.00028060557,0.00023065263,0.00019573854],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024432114,0.0007065682,0.00078056304,0.00033533893,0.00041581216,0.00058410293,0.0013822226,0.00080758636,0.0011471709],"category_scores_gemma":[0.0064883986,0.0004548474,0.00034490039,0.0002521557,0.0011834911,0.0012375119,0.0017375925,0.0016403666,0.00019264732],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009688806,0.000052846044,0.0010353386,0.00004738616,0.000035313078,0.00005362069,0.00008767552,0.9505045,0.0017870393,0.014014305,0.0006010257,0.03168403],"study_design_scores_gemma":[0.00000498784,0.000015713798,0.000034925743,0.0000026730872,0.0000021790113,0.0000053116696,0.0000033143795,0.99600536,0.00026447783,0.0035164629,0.0001426204,0.0000019950703],"about_ca_topic_score_codex":0.002227353,"about_ca_topic_score_gemma":0.003152945,"teacher_disagreement_score":0.0024432114,"about_ca_system_score_codex":0.00089257094,"about_ca_system_score_gemma":0.0013964635,"threshold_uncertainty_score":0.012921095},"labels":[],"label_agreement":null},{"id":"W4352994281","doi":"10.1007/978-3-031-25549-6_3","title":"Efficient Deep Reinforcement Learning via Policy-Extended Successor Feature Approximator","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Successor cardinal; Reinforcement learning; Computer science; Artificial intelligence; Generalization; Generalizability theory; Representation (politics); Decoupling (probability); Feature learning; Machine learning; Task (project management); Policy learning; Feature (linguistics); Mathematics","score_opus":0.013348182838903204,"score_gpt":0.25064532749151375,"score_spread":0.23729714465261054,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4352994281","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016185777,0.0003135743,0.97923136,0.00016941427,0.00008473618,0.00003313396,0.000044797893,0.0009460963,0.0029910468],"genre_scores_gemma":[0.82950586,0.00015793098,0.16381338,0.00013544672,0.00005221927,0.00012639274,0.0001278224,0.00012116683,0.0059597986],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996886,0.00006434457,0.000017515478,0.000073459065,0.000092152106,0.00006398274],"domain_scores_gemma":[0.9992582,0.00042990508,0.000056149816,0.000088719346,0.00012515622,0.00004181683],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008468799,0.0006441115,0.001236528,0.0002778211,0.0003027889,0.0007618179,0.001292194,0.0011199655,0.00412077],"category_scores_gemma":[0.002058628,0.0005064363,0.00035855675,0.00034804258,0.00063305936,0.0009186539,0.0012982578,0.0017528564,0.0006876234],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018338172,0.00012673586,0.00036638923,0.00007869209,0.000034435252,0.00007237656,0.0000358141,0.7883933,0.003542497,0.015219118,0.0028821835,0.1890651],"study_design_scores_gemma":[0.0000055903824,0.000010381875,0.000016909884,0.0000018173978,0.0000017811128,0.000005240022,9.71207e-7,0.99796367,0.0002319319,0.0016605465,0.00009987099,0.000001353177],"about_ca_topic_score_codex":0.004918677,"about_ca_topic_score_gemma":0.0059142555,"teacher_disagreement_score":0.004918677,"about_ca_system_score_codex":0.00084576505,"about_ca_system_score_gemma":0.0014356825,"threshold_uncertainty_score":0.013785362},"labels":[],"label_agreement":null},{"id":"W4353091694","doi":"10.1038/s41467-023-37180-x","title":"Catalyzing next-generation Artificial Intelligence through NeuroAI","year":2023,"lang":"en","type":"review","venue":"Nature Communications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":283,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute; Ontario Brain Institute; Montreal Neurological Institute and Hospital","funders":"National Institute of Neurological Disorders and Stroke; National Institute of General Medical Sciences; National Eye Institute; National Institute of Mental Health; Natural Sciences and Engineering Research Council of Canada; Intelligence Advanced Research Projects Activity; Defense Advanced Research Projects Agency; Office of Naval Research; Cold Spring Harbor Laboratory; Multidisciplinary University Research Initiative; James S. McDonnell Foundation; Canadian Institute for Advanced Research; National Science Foundation; Semiconductor Research Corporation; National Institutes of Health; Lourie Foundation; Howard Hughes Medical Institute","keywords":"Computer science; Artificial intelligence; Data science","score_opus":0.3747334748993592,"score_gpt":0.4284879568866756,"score_spread":0.053754481987316416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4353091694","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00024741222,0.9869784,0.0018814189,0.0013025646,0.00048923655,0.000008023307,0.000024327732,0.000035866728,0.009032658],"genre_scores_gemma":[0.003970777,0.99128526,0.001106015,0.0007732622,0.0003747559,0.0000216406,0.000051733798,0.00001093849,0.002405642],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.999777,0.00004886367,0.000015637754,0.000035522775,0.00009696603,0.000026020987],"domain_scores_gemma":[0.9996234,0.00022696349,0.00003227388,0.000022267881,0.000062751285,0.00003229309],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006930418,0.000734062,0.0006269018,0.0012576345,0.00031602298,0.001393294,0.0010722271,0.001498413,0.0052232966],"category_scores_gemma":[0.0010638381,0.0002627664,0.00043099484,0.001353084,0.00094793737,0.0027884343,0.0011434206,0.0026827445,0.003612658],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004503164,0.00006884576,0.00015402635,0.008527664,0.00008248502,0.00016188425,0.00011696471,0.0017115597,0.0017603677,0.10227857,0.04317736,0.8419152],"study_design_scores_gemma":[0.00000731559,0.000029461142,0.00014092411,0.0012245353,0.000024011912,0.00025130514,0.000031545762,0.00026376895,0.00042692464,0.02009925,0.9774889,0.0000120370205],"about_ca_topic_score_codex":0.0008005393,"about_ca_topic_score_gemma":0.0013057217,"teacher_disagreement_score":0.0052232966,"about_ca_system_score_codex":0.00085657503,"about_ca_system_score_gemma":0.0010201217,"threshold_uncertainty_score":0.017473698},"labels":[],"label_agreement":null},{"id":"W4360764167","doi":"10.1109/icmla55696.2022.00101","title":"Hyperparameter Tuning in Offline Reinforcement Learning","year":2022,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Lakehead University","funders":"","keywords":"Hyperparameter; Reinforcement learning; Benchmark (surveying); Metric (unit); Computer science; Machine learning; Performance metric; Artificial intelligence; Scheme (mathematics); Hyperparameter optimization; Friedman test; Statistical hypothesis testing; Statistics; Mathematics; Engineering","score_opus":0.01903993677232508,"score_gpt":0.2398741299863006,"score_spread":0.22083419321397552,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4360764167","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.046801377,0.0009445363,0.94321465,0.00044707034,0.0001388382,0.00019089505,0.00014804215,0.004732714,0.0033818053],"genre_scores_gemma":[0.85021544,0.0001683484,0.14579734,0.0005164733,0.000078083256,0.00041817344,0.00034204256,0.00059453025,0.0018696018],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99681646,0.0014875846,0.0001782501,0.0006818242,0.0005320002,0.00030386582],"domain_scores_gemma":[0.99323046,0.003641822,0.0006293257,0.0014149988,0.00078807335,0.00029528383],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043882052,0.0021758303,0.0017634094,0.00083220674,0.0005792055,0.0017772195,0.0027385433,0.0019279327,0.0030741098],"category_scores_gemma":[0.023487575,0.00067921437,0.00057685585,0.00045479322,0.0019281317,0.0024409322,0.0022318338,0.003983528,0.0013973016],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037930062,0.00026255325,0.0025771467,0.00015952568,0.00009190156,0.00010973862,0.00011339218,0.8621983,0.004386362,0.006520212,0.0038211788,0.119380526],"study_design_scores_gemma":[0.000051864834,0.000095371644,0.00015182536,0.00002008439,0.000009381119,0.000029842178,0.0000151296135,0.9909729,0.001710642,0.0062416177,0.0006871898,0.000014342637],"about_ca_topic_score_codex":0.0021563207,"about_ca_topic_score_gemma":0.002477139,"teacher_disagreement_score":0.0043882052,"about_ca_system_score_codex":0.0012775704,"about_ca_system_score_gemma":0.0018473347,"threshold_uncertainty_score":0.023207307},"labels":[],"label_agreement":null},{"id":"W4360764815","doi":"10.1109/icmla55696.2022.00044","title":"Benchmarking Offline Reinforcement Learning","year":2022,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Lakehead University","funders":"","keywords":"Benchmarking; Reinforcement learning; Computer science; Hyperparameter; Machine learning; Artificial intelligence; Offline learning; Performance improvement; Online learning; Engineering","score_opus":0.013732024615425604,"score_gpt":0.22941801827976827,"score_spread":0.21568599366434266,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4360764815","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.29190403,0.006273308,0.5814791,0.0011920454,0.0014953339,0.0007001641,0.011059308,0.06748007,0.03841674],"genre_scores_gemma":[0.79779994,0.0007174527,0.17614473,0.0004826491,0.00012218488,0.00062315987,0.0162881,0.0019991251,0.0058226623],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9961888,0.0012808179,0.00026246774,0.0010238142,0.00082401786,0.0004200916],"domain_scores_gemma":[0.9925673,0.003942952,0.0002868081,0.0018367126,0.0010693504,0.0002968086],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034632953,0.0022176153,0.0017351666,0.0010913372,0.00063128135,0.0017295971,0.0032557778,0.002070485,0.008192163],"category_scores_gemma":[0.014763648,0.0005832271,0.0008809311,0.0010911246,0.0012272624,0.002030501,0.0018067067,0.0026035963,0.0036252367],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009297007,0.0007919881,0.004031412,0.00075085607,0.0002498519,0.00010762601,0.00006416783,0.80625397,0.0025106254,0.00493269,0.0201417,0.15923545],"study_design_scores_gemma":[0.00015077884,0.00028540136,0.00097524974,0.000044447763,0.000021578295,0.000048253984,0.000026908381,0.9839103,0.0059137866,0.0034125207,0.0051843133,0.000026431851],"about_ca_topic_score_codex":0.009267435,"about_ca_topic_score_gemma":0.008013245,"teacher_disagreement_score":0.009267435,"about_ca_system_score_codex":0.0015320851,"about_ca_system_score_gemma":0.0022101158,"threshold_uncertainty_score":0.0274055},"labels":[],"label_agreement":null},{"id":"W4360764838","doi":"10.1109/icmla55696.2022.00097","title":"IGN : Implicit Generative Networks","year":2022,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Science North","funders":"","keywords":"Discriminator; Reinforcement learning; Computer science; Quantile regression; Generator (circuit theory); Baseline (sea); Quantile; Artificial intelligence; Bellman equation; State (computer science); Generative grammar; Action (physics); Machine learning; Mathematical optimization; Algorithm; Econometrics; Mathematics; Power (physics)","score_opus":0.013866665707804763,"score_gpt":0.23323073197724947,"score_spread":0.2193640662694447,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4360764838","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010486973,0.0005398671,0.9726627,0.0007576132,0.00014547964,0.000053786007,0.00062777504,0.0032109222,0.011514919],"genre_scores_gemma":[0.7188214,0.00078985485,0.24699157,0.0010632505,0.00019697986,0.00031916532,0.0028380288,0.0013937312,0.02758598],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995227,0.00016742556,0.000015174683,0.0001386833,0.000102781334,0.00005324176],"domain_scores_gemma":[0.99894947,0.0006613822,0.000050853647,0.00018307165,0.000096382035,0.000058946494],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00081075344,0.00091323827,0.0008738676,0.0005617919,0.00046088372,0.0011044008,0.0020241463,0.0013151799,0.010891889],"category_scores_gemma":[0.0045760833,0.0005557061,0.00079720456,0.00065338204,0.0011516217,0.0019438457,0.0021855296,0.0027355119,0.002891688],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011369657,0.00008832786,0.0013908905,0.00013040549,0.000070925795,0.00020468215,0.00010813954,0.7102369,0.0014880807,0.16377996,0.011844225,0.11054379],"study_design_scores_gemma":[0.000009652308,0.000009335695,0.00006333154,0.000010410172,0.000005514683,0.0000321057,0.000004301508,0.9273109,0.00029422637,0.06981039,0.002445039,0.0000047904673],"about_ca_topic_score_codex":0.0037502972,"about_ca_topic_score_gemma":0.0077997283,"teacher_disagreement_score":0.010891889,"about_ca_system_score_codex":0.0010984788,"about_ca_system_score_gemma":0.00088792044,"threshold_uncertainty_score":0.036436975},"labels":[],"label_agreement":null},{"id":"W4360771686","doi":"10.1109/icmla55696.2022.00008","title":"Keynotes","year":2022,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Microsoft (Canada)","funders":"","keywords":"Computer science","score_opus":0.010275642458580992,"score_gpt":0.20805035321039803,"score_spread":0.19777471075181705,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4360771686","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010256135,0.00558693,0.0042071957,0.12148083,0.1794347,0.00052966614,0.005779024,0.0015241173,0.68043184],"genre_scores_gemma":[0.008627708,0.0020835116,0.0009762573,0.028193656,0.013289939,0.0003310406,0.002239938,0.00060322136,0.9436547],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99830025,0.0003080117,0.00012619948,0.00034271693,0.00068041444,0.00024254169],"domain_scores_gemma":[0.9945134,0.00151871,0.00020901141,0.00044015268,0.0024082914,0.0009103605],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0019182842,0.00096748315,0.00086342386,0.0016889899,0.003336485,0.004109828,0.0016615617,0.0035323328,0.58612967],"category_scores_gemma":[0.015124276,0.0003634868,0.00070387824,0.0014013394,0.00082157255,0.0038771403,0.003069698,0.0051079905,0.34174246],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000028601304,0.00001232889,0.00005516548,0.00004304909,0.0000013235605,0.000036133166,0.00005450906,0.000017778373,0.00006977834,0.005002295,0.9827056,0.011973464],"study_design_scores_gemma":[0.0000045231054,0.0000064894757,0.00010134639,0.00004171646,0.0000012823962,0.000035726818,0.00008688919,0.000019324394,0.000051580297,0.0010115434,0.9986362,0.0000034627737],"about_ca_topic_score_codex":0.0050216406,"about_ca_topic_score_gemma":0.007042993,"teacher_disagreement_score":0.41387033,"about_ca_system_score_codex":0.0029444501,"about_ca_system_score_gemma":0.0019865506,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W4361273506","doi":"10.3390/systems11040180","title":"It’s All about Reward: Contrasting Joint Rewards and Individual Reward in Centralized Learning Decentralized Execution Algorithms","year":2023,"lang":"en","type":"article","venue":"Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Mitacs","keywords":"Reinforcement learning; Variance (accounting); Computer science; Joint (building); Function (biology); Foraging; Machine learning; Artificial intelligence; Value (mathematics); Engineering","score_opus":0.06219908550440429,"score_gpt":0.2892080950814729,"score_spread":0.2270090095770686,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4361273506","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27425453,0.0010071481,0.71544164,0.0017197867,0.00009896574,0.000105546394,0.00002569032,0.00024635709,0.007100307],"genre_scores_gemma":[0.96378464,0.00013310641,0.035320606,0.00011070833,0.000035433237,0.00003790878,0.0000110857245,0.00003883279,0.00052772096],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9963246,0.0022975847,0.00011548078,0.00039941288,0.000602293,0.00026068013],"domain_scores_gemma":[0.9661827,0.02850111,0.0018522277,0.0012397022,0.0013343456,0.00088997907],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007112126,0.0005548114,0.0012804022,0.00044886785,0.00059141853,0.0017557223,0.0009912825,0.0012309309,0.001057821],"category_scores_gemma":[0.037763968,0.00028050554,0.00034507352,0.0004463707,0.0021736203,0.0031110384,0.0019707014,0.0018608937,0.0001479448],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007374689,0.00032481525,0.0049213404,0.0001978694,0.00011304815,0.00007088216,0.00031984932,0.8310218,0.0026313988,0.0733522,0.000654518,0.08565481],"study_design_scores_gemma":[0.00006459595,0.00039006732,0.0013512699,0.00003143322,0.000025848198,0.000025547619,0.00007087715,0.9444976,0.0011105554,0.052013505,0.00039806942,0.000020656129],"about_ca_topic_score_codex":0.0015846051,"about_ca_topic_score_gemma":0.0015671313,"teacher_disagreement_score":0.007112126,"about_ca_system_score_codex":0.0013246981,"about_ca_system_score_gemma":0.0013115086,"threshold_uncertainty_score":0.037612975},"labels":[],"label_agreement":null},{"id":"W4364304082","doi":"10.1109/raai56146.2022.10092979","title":"Addressing Different Goal Selection Strategies In Hindsight Experience Replay With Actor-Critic Methods For Robotic Hand Manipulation","year":2022,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Hindsight bias; Reinforcement learning; Computer science; Task (project management); Artificial intelligence; Selection (genetic algorithm); Robot; Block (permutation group theory); Binary number; Machine learning; Human–computer interaction; Engineering; Mathematics","score_opus":0.07996687853250528,"score_gpt":0.373707120529135,"score_spread":0.29374024199662974,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4364304082","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.099614285,0.00042423114,0.89630795,0.00021554274,0.000049453614,0.000060411177,0.000023331733,0.0009850294,0.0023198149],"genre_scores_gemma":[0.923681,0.00008398636,0.07415709,0.00010226575,0.000013737829,0.00007132263,0.00003316842,0.00007389468,0.0017834255],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99968064,0.00013525391,0.000017772529,0.00006912813,0.00005652306,0.000040775525],"domain_scores_gemma":[0.9987986,0.00074966456,0.00013633298,0.000108001215,0.0001181262,0.00008925907],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011884267,0.000874077,0.0005623295,0.00018987559,0.0002067587,0.00048457299,0.0008350508,0.0007545225,0.001245845],"category_scores_gemma":[0.003772844,0.0003545614,0.00029483566,0.000120878576,0.0006773521,0.0006531021,0.00096450397,0.0013400051,0.0002392168],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002292448,0.00009841019,0.0008996407,0.00007604609,0.00005004672,0.00009011717,0.000114730836,0.9381865,0.0062786485,0.003221983,0.0005950988,0.05015957],"study_design_scores_gemma":[0.000011055043,0.00005785179,0.00008354646,0.000004582972,0.0000034025093,0.00001104462,0.0000054979823,0.9975183,0.00091513887,0.0012273955,0.0001582068,0.0000039303027],"about_ca_topic_score_codex":0.0023426174,"about_ca_topic_score_gemma":0.0026977614,"teacher_disagreement_score":0.0023426174,"about_ca_system_score_codex":0.00046469382,"about_ca_system_score_gemma":0.0007138753,"threshold_uncertainty_score":0.0062850714},"labels":[],"label_agreement":null},{"id":"W4367016230","doi":"10.1109/tse.2023.3269804","title":"A Search-Based Testing Approach for Deep Reinforcement Learning Agents","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Cisco Systems (Canada); École de Technologie Supérieure; University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Reinforcement learning; Artificial intelligence; Machine learning; Software testing; Software engineering; Programming language; Software","score_opus":0.04544628432889011,"score_gpt":0.25729172732987793,"score_spread":0.21184544300098782,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4367016230","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04858564,0.00032042782,0.9451978,0.0005923502,0.000056740773,0.00015587345,0.00006094163,0.0017000509,0.0033301448],"genre_scores_gemma":[0.8344012,0.00010196091,0.16252422,0.00034652906,0.000031115305,0.00023597223,0.00010533221,0.00013091254,0.002122773],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99802303,0.0007338246,0.0001201682,0.0003591236,0.00046430144,0.0002995104],"domain_scores_gemma":[0.9933891,0.004184209,0.00055726833,0.00046234977,0.0010059758,0.00040100692],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027163944,0.0011532074,0.0013423503,0.000852665,0.00058863097,0.0010081884,0.0028740403,0.0017850004,0.0026132977],"category_scores_gemma":[0.011490425,0.00064077426,0.00082829257,0.0004088538,0.0020611372,0.0017438735,0.0018497461,0.0018082209,0.00036773641],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016019838,0.00014195862,0.0019195474,0.00008650802,0.000052684674,0.0001695513,0.000101372396,0.92903984,0.0024764563,0.017337697,0.0008314451,0.04768268],"study_design_scores_gemma":[0.000014720185,0.000038031856,0.000046356614,0.000004846043,0.00000452049,0.000011139421,0.0000060290254,0.99540675,0.00037653462,0.0039166776,0.00017091591,0.0000035906353],"about_ca_topic_score_codex":0.006458457,"about_ca_topic_score_gemma":0.0049098185,"teacher_disagreement_score":0.006458457,"about_ca_system_score_codex":0.0015694884,"about_ca_system_score_gemma":0.0026025602,"threshold_uncertainty_score":0.014365852},"labels":[],"label_agreement":null},{"id":"W4367060663","doi":"10.48550/arxiv.2304.12567","title":"Proto-Value Networks: Scaling Representation Learning with Auxiliary Tasks","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Computer science; Successor cardinal; Representation (politics); Value (mathematics); Artificial intelligence; Function (biology); Carry (investment); Feature learning; Object (grammar); Machine learning; Mathematics","score_opus":0.09279756993421001,"score_gpt":0.21577094032397215,"score_spread":0.12297337038976214,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4367060663","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.058526672,0.00017883051,0.9366651,0.00034604702,0.0000554563,0.00006926137,0.000060377602,0.0005589741,0.0035392107],"genre_scores_gemma":[0.78145945,0.00021613193,0.21447271,0.00021562238,0.00005119957,0.00021587258,0.00012729027,0.00012889148,0.0031126693],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99960154,0.00015470645,0.000017746615,0.0000984686,0.00008138451,0.000046134934],"domain_scores_gemma":[0.9976943,0.0013923428,0.00022831674,0.00030752044,0.0001994767,0.00017803622],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001620883,0.00083206437,0.00077192334,0.00042646943,0.00035863608,0.000948446,0.0015003543,0.0011622715,0.0034871008],"category_scores_gemma":[0.008483458,0.00039693073,0.00041874257,0.00040342056,0.001312957,0.0029628149,0.001963267,0.001960412,0.00046653266],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026372136,0.00023393784,0.0013496875,0.00013068983,0.000053429812,0.00010171622,0.00015042149,0.74760914,0.0055566304,0.09947185,0.0026501552,0.14242856],"study_design_scores_gemma":[0.000014563346,0.000048083682,0.00006587854,0.00000726178,0.0000042678075,0.000014017151,0.0000055411547,0.96821356,0.00081399473,0.030406283,0.00040253255,0.0000040173177],"about_ca_topic_score_codex":0.0009321825,"about_ca_topic_score_gemma":0.00094688125,"teacher_disagreement_score":0.0034871008,"about_ca_system_score_codex":0.00084798335,"about_ca_system_score_gemma":0.00070874445,"threshold_uncertainty_score":0.011665523},"labels":[],"label_agreement":null},{"id":"W4367189765","doi":"10.48550/arxiv.2304.13223","title":"Reinforcement Learning with Partial Parametric Model Knowledge","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; Honeywell (Canada)","funders":"","keywords":"Reinforcement learning; Computer science; Ignorance; Bridge (graph theory); Parametric statistics; Control (management); Complete information; Linear-quadratic regulator; Mathematical optimization; Artificial intelligence; Mathematics; Mathematical economics; Law","score_opus":0.1275187607125167,"score_gpt":0.21539233160582583,"score_spread":0.08787357089330913,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4367189765","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009320205,0.0001394915,0.9884235,0.00019145869,0.000028040282,0.000021595462,0.000023139046,0.0003142533,0.0015383082],"genre_scores_gemma":[0.89240664,0.00019532756,0.10409764,0.00018257911,0.000066401044,0.00014368378,0.00008510384,0.00007178871,0.0027508207],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99884546,0.00043507275,0.000046581838,0.00024166859,0.00032149194,0.00010975076],"domain_scores_gemma":[0.99630415,0.002337651,0.0003911293,0.0004908048,0.00032555868,0.00015079061],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017785059,0.001091645,0.0012602222,0.00040418585,0.00031480903,0.001178108,0.0017583853,0.0010472997,0.0018879906],"category_scores_gemma":[0.0073450613,0.000537941,0.00074307487,0.0004589151,0.0017013662,0.001928006,0.0018162384,0.0019446467,0.0003847866],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006685634,0.00005074746,0.00033328406,0.000069924856,0.000048989048,0.000079831945,0.000060404385,0.94938874,0.0010494292,0.018955996,0.0005188607,0.029376945],"study_design_scores_gemma":[0.000011057067,0.000033324333,0.000034324927,0.000004814686,0.0000055501214,0.000009413384,0.000002884089,0.9898302,0.00031714488,0.009475103,0.00027051946,0.0000056720155],"about_ca_topic_score_codex":0.002774829,"about_ca_topic_score_gemma":0.0021389571,"teacher_disagreement_score":0.002774829,"about_ca_system_score_codex":0.00064709445,"about_ca_system_score_gemma":0.001286333,"threshold_uncertainty_score":0.009405792},"labels":[],"label_agreement":null},{"id":"W4367319168","doi":"10.1145/3543873.3587661","title":"Investigating Action-Space Generalization in Reinforcement Learning for Recommendation Systems","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Generalization; Reinforcement learning; Computer science; Action (physics); Space (punctuation); Artificial intelligence; Machine learning; Mathematics","score_opus":0.0737814705673724,"score_gpt":0.31667723497223127,"score_spread":0.24289576440485888,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4367319168","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11994379,0.0011096424,0.87232393,0.0020611545,0.00007886307,0.00016217417,0.000117857584,0.00040818995,0.0037945048],"genre_scores_gemma":[0.9529475,0.000512352,0.043661244,0.00036981577,0.00007825062,0.00019370207,0.00013617314,0.00007356871,0.0020274334],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99703145,0.0015830093,0.00013402557,0.000613171,0.00035832947,0.0002798997],"domain_scores_gemma":[0.9638111,0.030715939,0.0015748381,0.0017149365,0.0014855941,0.0006975327],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008401311,0.0011912396,0.0022708543,0.0006743823,0.000712719,0.001441087,0.0019790952,0.002217666,0.0024955962],"category_scores_gemma":[0.049430754,0.0008104236,0.0012124967,0.00068074634,0.0028393783,0.003638553,0.0019347608,0.0037498178,0.00026783446],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009339934,0.000113330956,0.0025812585,0.000118765965,0.00008910341,0.00008491643,0.00023907014,0.9361517,0.00062277005,0.044570528,0.00061535614,0.014719811],"study_design_scores_gemma":[0.000010700189,0.000031952506,0.00020088613,0.0000080563395,0.000005532602,0.000008223138,0.000011963169,0.98206156,0.00006482918,0.017463781,0.00012615895,0.000006344339],"about_ca_topic_score_codex":0.013992537,"about_ca_topic_score_gemma":0.0085403165,"teacher_disagreement_score":0.013992537,"about_ca_system_score_codex":0.0028787258,"about_ca_system_score_gemma":0.001798869,"threshold_uncertainty_score":0.044430852},"labels":[],"label_agreement":null},{"id":"W4368227480","doi":"10.1109/iccae56788.2023.10111485","title":"Proximity-Based Reward System and Reinforcement Learning for Path Planning","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Moncton","funders":"","keywords":"Reinforcement learning; Motion planning; Computer science; Artificial intelligence; Path (computing); Automation; Machine learning; Robotics; Field (mathematics); Task (project management); Robot; Engineering; Mathematics","score_opus":0.03200085973623594,"score_gpt":0.26905691784965347,"score_spread":0.2370560581134175,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4368227480","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011828658,0.00042247592,0.98541576,0.00018240363,0.00005825546,0.00004789524,0.000019028239,0.00025924065,0.0017662332],"genre_scores_gemma":[0.787259,0.00042338992,0.20788601,0.00014221332,0.00009768932,0.00020178268,0.000043476935,0.00007003594,0.0038764374],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998765,0.0005077087,0.00007284729,0.00024002597,0.0003192553,0.000095265],"domain_scores_gemma":[0.99758625,0.0014127078,0.00027615935,0.00016649085,0.0003885256,0.00016994709],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012980032,0.0005589344,0.000845558,0.0005303166,0.0005119294,0.0008044779,0.0009245741,0.0009280823,0.002945954],"category_scores_gemma":[0.006001695,0.0002794128,0.00048436772,0.00049384806,0.0013723412,0.0016809626,0.0013533159,0.0015534016,0.00039089666],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022427211,0.00010987474,0.00063689274,0.0001758124,0.000049915896,0.000106093736,0.00012335466,0.8412043,0.0048744464,0.064815626,0.001043487,0.086635925],"study_design_scores_gemma":[0.000025424712,0.00012349884,0.0001877217,0.000010208826,0.000010252585,0.000049454284,0.000007658866,0.98089325,0.0011008753,0.016608326,0.00096593634,0.000017379069],"about_ca_topic_score_codex":0.002265876,"about_ca_topic_score_gemma":0.0012589847,"teacher_disagreement_score":0.002945954,"about_ca_system_score_codex":0.0011397213,"about_ca_system_score_gemma":0.0009330289,"threshold_uncertainty_score":0.009855211},"labels":[],"label_agreement":null},{"id":"W4372259970","doi":"10.1109/icassp49357.2023.10095236","title":"MEET: A Monte Carlo Exploration-Exploitation Trade-Off for Buffer Sampling","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Infineon Technologies (Canada)","funders":"","keywords":"Reinforcement learning; Computer science; Sampling (signal processing); Convergence (economics); Task (project management); Importance sampling; Bellman equation; Selection (genetic algorithm); Monte Carlo method; Function (biology); Machine learning; State (computer science); Mathematical optimization; Artificial intelligence; Algorithm; Statistics; Mathematics; Engineering","score_opus":0.08781016052838872,"score_gpt":0.3141202326013707,"score_spread":0.22631007207298195,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4372259970","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024272673,0.00043082048,0.97208786,0.0002599408,0.000059168964,0.00014908677,0.000042902604,0.00089801225,0.0017995762],"genre_scores_gemma":[0.7653271,0.00017651192,0.23099901,0.00028498436,0.00008664206,0.00038901626,0.00012798175,0.00023458703,0.0023740865],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99847,0.0006171169,0.00007979794,0.00024941604,0.00041273722,0.00017090351],"domain_scores_gemma":[0.9944707,0.0037901385,0.0003666701,0.00046346846,0.00045774665,0.00045131485],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003897645,0.001209816,0.0015452758,0.000883231,0.0006398046,0.0013168405,0.002673821,0.0019626066,0.0033407153],"category_scores_gemma":[0.013998433,0.00071839767,0.00066944043,0.00050445204,0.0011809696,0.0024912946,0.0022621222,0.0020536967,0.0005977015],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00085216505,0.00032082884,0.0024586604,0.00016659738,0.00012051786,0.0001115712,0.00018123684,0.81885326,0.004111227,0.024661083,0.0021995122,0.1459633],"study_design_scores_gemma":[0.000028234575,0.00009004578,0.000085992295,0.000008940764,0.0000076230626,0.000020856329,0.000007575622,0.9945627,0.0008296265,0.0040690373,0.00028224735,0.0000070399346],"about_ca_topic_score_codex":0.0019426616,"about_ca_topic_score_gemma":0.00241481,"teacher_disagreement_score":0.003897645,"about_ca_system_score_codex":0.0010698192,"about_ca_system_score_gemma":0.0017188616,"threshold_uncertainty_score":0.020613015},"labels":[],"label_agreement":null},{"id":"W4375798913","doi":"10.1109/lra.2023.3273421","title":"Multi-Abstractive Neural Controller: An Efficient Hierarchical Control Architecture for Interactive Driving","year":2023,"lang":"en","type":"article","venue":"IEEE Robotics and Automation Letters","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Toyota Research Institute","keywords":"Interpretability; Controller (irrigation); Computer science; Artificial neural network; Artificial intelligence; Set (abstract data type); Machine learning; Programming language","score_opus":0.014628632478758305,"score_gpt":0.2679701900467087,"score_spread":0.2533415575679504,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4375798913","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036267474,0.000115098046,0.95611674,0.00015552047,0.000045654913,0.00006471586,0.00004724502,0.0023052618,0.0048822183],"genre_scores_gemma":[0.82443786,0.000070154194,0.17157434,0.000113964394,0.000015412214,0.00010055337,0.00008627403,0.000075703014,0.0035256848],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99985516,0.000017116272,0.0000072171924,0.00004686114,0.000051144016,0.000022480383],"domain_scores_gemma":[0.99989605,0.000022054213,0.000014319561,0.000027559487,0.000024827239,0.000015241291],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00028536824,0.00032024452,0.00019556709,0.00018861734,0.00024838018,0.00043880625,0.0015349808,0.00053388235,0.0021623415],"category_scores_gemma":[0.00042294935,0.0001879255,0.00031036127,0.00012909019,0.00050889235,0.0005252025,0.0007197062,0.00071696716,0.0003645548],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016936655,0.00018622453,0.0008391112,0.00011117218,0.000055935434,0.00022192893,0.00025906376,0.62988114,0.06606954,0.036157247,0.0028400375,0.26320925],"study_design_scores_gemma":[0.000008601317,0.000037556612,0.000113787835,0.0000031558395,0.0000053649987,0.000018168068,0.0000040828227,0.99240535,0.0034107913,0.002887863,0.0011002306,0.0000049789337],"about_ca_topic_score_codex":0.0049958997,"about_ca_topic_score_gemma":0.00797506,"teacher_disagreement_score":0.0049958997,"about_ca_system_score_codex":0.0006896221,"about_ca_system_score_gemma":0.00056450785,"threshold_uncertainty_score":0.0099336505},"labels":[],"label_agreement":null},{"id":"W4376223516","doi":"10.1613/jair.1.14580","title":"Exploiting Action Impact Regularity and Exogenous State Variables for Offline Reinforcement Learning","year":2023,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Component (thermodynamics); Exploit; Property (philosophy); Offline learning; Key (lock); Class (philosophy); Artificial intelligence; Machine learning; Action (physics); State (computer science); Reinforcement; Online and offline; State action; Online learning; Algorithm","score_opus":0.29473503459567824,"score_gpt":0.4630904091938908,"score_spread":0.16835537459821254,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4376223516","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026665404,0.000091942304,0.97110176,0.00021219953,0.000019933492,0.000077193086,0.00004024864,0.00042729816,0.0013640393],"genre_scores_gemma":[0.8793874,0.00010318866,0.11859851,0.00018723976,0.000036865615,0.00019903341,0.00013409181,0.000117632815,0.0012360499],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988194,0.00043074507,0.00007712655,0.0002884094,0.00022592525,0.00015832315],"domain_scores_gemma":[0.9845781,0.012079989,0.0010659931,0.0013223892,0.0005153754,0.00043826696],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036541633,0.0010889128,0.0016060722,0.0005058318,0.00048757446,0.0010144482,0.0015122066,0.0013156427,0.002520514],"category_scores_gemma":[0.018113796,0.0007169983,0.0006300849,0.00047161605,0.0021672929,0.0023414341,0.0018981466,0.003248899,0.00039274522],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014581661,0.000119279975,0.0015529818,0.000060641687,0.000029296874,0.000070271504,0.000067467656,0.9516208,0.0009576736,0.020273777,0.00046393438,0.024637902],"study_design_scores_gemma":[0.000013771538,0.000029792713,0.00007534039,0.0000048291085,0.0000026081736,0.000008989831,0.0000040038453,0.9908174,0.00029890315,0.008624328,0.00011686321,0.0000032378923],"about_ca_topic_score_codex":0.0035553055,"about_ca_topic_score_gemma":0.0032931813,"teacher_disagreement_score":0.0036541633,"about_ca_system_score_codex":0.0011636984,"about_ca_system_score_gemma":0.0025856872,"threshold_uncertainty_score":0.019325316},"labels":[],"label_agreement":null},{"id":"W4376606197","doi":"10.1109/aero55745.2023.10115987","title":"Stress Propagation in Human-Robot Teams Based on Computational Logic Model","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Robot; Stress (linguistics); Toll; Core (optical fiber); Influence diagram; Artificial intelligence; Human–computer interaction; Machine learning; Decision tree","score_opus":0.04069771383618445,"score_gpt":0.29980528452134625,"score_spread":0.25910757068516177,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4376606197","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.124036714,0.00015790628,0.8597327,0.0007410153,0.00005837336,0.00009628684,0.0002398419,0.00026562842,0.014671572],"genre_scores_gemma":[0.9474709,0.00013581161,0.04690527,0.00009114657,0.000027394895,0.0001626367,0.00011334777,0.000027404667,0.005065988],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993649,0.00020847966,0.00003252061,0.00014701636,0.00014440813,0.00010272817],"domain_scores_gemma":[0.99854076,0.0008966627,0.00019417693,0.000082591716,0.00014352679,0.00014228985],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00086663436,0.0005226082,0.0006588575,0.00054609374,0.0006690316,0.0014804883,0.0014039997,0.0008311476,0.0038100681],"category_scores_gemma":[0.002605417,0.00030018762,0.00084548513,0.00040275042,0.0012713918,0.0016046361,0.0011833875,0.0010914047,0.00034219865],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007454884,0.00006164199,0.0014157973,0.000034192846,0.000030467136,0.00020631313,0.00018098974,0.88336843,0.0008460846,0.10756191,0.0004897662,0.0057298355],"study_design_scores_gemma":[0.000008073925,0.000014458013,0.00008139155,0.0000025942786,0.000007379144,0.000012631699,0.000015593427,0.9745959,0.00010041939,0.024959609,0.000197784,0.000004104622],"about_ca_topic_score_codex":0.00651887,"about_ca_topic_score_gemma":0.00411685,"teacher_disagreement_score":0.00651887,"about_ca_system_score_codex":0.0014571117,"about_ca_system_score_gemma":0.0012469087,"threshold_uncertainty_score":0.0129618645},"labels":[],"label_agreement":null},{"id":"W4376639574","doi":"10.1145/3564246.3585161","title":"Optimistic MLE: A Generic Model-Based Algorithm for Partially Observable Sequential Decision Making","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Google; National Science Foundation","keywords":"Observable; Computer science; Simple (philosophy); Class (philosophy); Reinforcement learning; Rank (graph theory); Algorithm; Mathematical optimization; Artificial intelligence; Mathematics","score_opus":0.07463704283305572,"score_gpt":0.31685020754951426,"score_spread":0.24221316471645854,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4376639574","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0017581328,0.00010292339,0.99601364,0.00017176743,0.000021696287,0.0000344108,0.000054399472,0.0007777968,0.0010652789],"genre_scores_gemma":[0.28997618,0.00029422593,0.70469934,0.00037467308,0.00009481724,0.00041207767,0.00034517623,0.000344369,0.0034591488],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99878424,0.0003876346,0.00006120263,0.0002289927,0.0003790008,0.00015898103],"domain_scores_gemma":[0.997884,0.0013161777,0.00017593893,0.00034900854,0.00017320218,0.0001016636],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022785997,0.0012277366,0.0014642105,0.00070744817,0.00051838794,0.001475938,0.0025775232,0.0016157196,0.0059179766],"category_scores_gemma":[0.00765771,0.00078672904,0.0009412079,0.00080596923,0.0011460405,0.0024041277,0.0034915793,0.0031533977,0.0013430932],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000227159,0.000093483155,0.0006620148,0.00017346037,0.00008526004,0.00009737059,0.00011400876,0.7502735,0.0015027644,0.089631744,0.0047785314,0.15236063],"study_design_scores_gemma":[0.000025775083,0.000022081527,0.000029907318,0.000012867309,0.0000063623916,0.00002133132,0.0000064945148,0.9624474,0.0004737214,0.03577906,0.0011678672,0.00000712847],"about_ca_topic_score_codex":0.001598687,"about_ca_topic_score_gemma":0.0024213018,"teacher_disagreement_score":0.0059179766,"about_ca_system_score_codex":0.0011222,"about_ca_system_score_gemma":0.0023873877,"threshold_uncertainty_score":0.019797564},"labels":[],"label_agreement":null},{"id":"W4377018773","doi":"10.32473/flairs.36.133317","title":"Evaluation of Techniques for Sim2Real Reinforcement Learning","year":2023,"lang":"en","type":"article","venue":"Proceedings of the ... International Florida Artificial Intelligence Research Society Conference","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Reinforcement learning; Computer science; Bridging (networking); Bridge (graph theory); Generalization; Noise (video); Domain (mathematical analysis); Human–computer interaction; Transfer of learning; Process (computing); Artificial intelligence; Mathematics","score_opus":0.2601722272290781,"score_gpt":0.42715826990483485,"score_spread":0.16698604267575673,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4377018773","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28570917,0.001009764,0.69193,0.00055042043,0.00024610548,0.0007466574,0.00031607202,0.007504613,0.0119871795],"genre_scores_gemma":[0.7585954,0.0001719403,0.23834252,0.00012472138,0.00002157539,0.0003549735,0.00036227365,0.00022856508,0.0017979699],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99697554,0.0013449808,0.0002009855,0.00047190825,0.00074077543,0.0002658289],"domain_scores_gemma":[0.99158186,0.0051682894,0.0004437659,0.0012064762,0.001310241,0.0002894505],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005848438,0.00120764,0.0007687734,0.00082037365,0.0005390255,0.00089359586,0.0026242102,0.0014371161,0.002931347],"category_scores_gemma":[0.012986366,0.00041501303,0.0005343151,0.0005189807,0.0011055176,0.0013922228,0.0017013825,0.0014112104,0.00063706643],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00088393193,0.000872548,0.0027906387,0.00042231663,0.0001114798,0.00009392251,0.0001612733,0.84142005,0.0042995983,0.00547519,0.0015261035,0.14194289],"study_design_scores_gemma":[0.00006205021,0.00023691262,0.00025477904,0.000009954087,0.0000074194472,0.000022071557,0.000022364935,0.9940376,0.003944613,0.0006585622,0.00073631137,0.0000074012996],"about_ca_topic_score_codex":0.004506418,"about_ca_topic_score_gemma":0.0037621327,"teacher_disagreement_score":0.005848438,"about_ca_system_score_codex":0.001959383,"about_ca_system_score_gemma":0.0013155086,"threshold_uncertainty_score":0.030929863},"labels":[],"label_agreement":null},{"id":"W4377971347","doi":"10.1109/tai.2023.3279057","title":"Interpreting Tangled Program Graphs Under Partially Observable Dota 2 Invoker Tasks","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Interpretability; Artificial intelligence; Graph; Context (archaeology); Machine learning; Task (project management); Theoretical computer science","score_opus":0.08019492026766367,"score_gpt":0.32383461407993724,"score_spread":0.24363969381227357,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4377971347","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12139292,0.00009171293,0.86868936,0.00037000465,0.000029165478,0.00013753845,0.0004036624,0.002562775,0.0063228933],"genre_scores_gemma":[0.78062534,0.00014103437,0.21339317,0.00014416728,0.000016272566,0.00023383461,0.0006097201,0.00046231964,0.0043741316],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991906,0.0002948894,0.000057649417,0.00019782461,0.0001646011,0.0000944065],"domain_scores_gemma":[0.9972862,0.0014640259,0.00033384754,0.00053209,0.00024687775,0.00013694138],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00088795344,0.0006710338,0.00032064994,0.0006684771,0.00042133304,0.001586246,0.0010198045,0.0008848854,0.003179056],"category_scores_gemma":[0.006252816,0.00040023337,0.0007991205,0.0002604088,0.001992469,0.002800152,0.002137968,0.0013536093,0.00031113243],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018665609,0.00012537785,0.003395136,0.00017973014,0.000044512217,0.0010084112,0.001745818,0.67276293,0.010479013,0.266059,0.0010542208,0.04295916],"study_design_scores_gemma":[0.00002087235,0.00005182957,0.0002903024,0.000025095836,0.000018455461,0.00006734954,0.00018392621,0.7853457,0.0048692743,0.20606935,0.003039346,0.000018408213],"about_ca_topic_score_codex":0.005632783,"about_ca_topic_score_gemma":0.007143462,"teacher_disagreement_score":0.005632783,"about_ca_system_score_codex":0.0013765483,"about_ca_system_score_gemma":0.0010254399,"threshold_uncertainty_score":0.011200011},"labels":[],"label_agreement":null},{"id":"W4378192263","doi":"10.1109/syscon53073.2023.10131174","title":"A Deep Reinforcement Learning Solution for the Low Level Motion Control of a Robot Manipulator System","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Reinforcement learning; Computer science; Robot; Artificial intelligence; Motion (physics); Collision avoidance; Object (grammar); Motion control; Tower; Motion planning; Robot control; Variety (cybernetics); Manipulator (device); Artificial neural network; Control (management); Simulation; Collision; Control engineering; Engineering; Mobile robot","score_opus":0.04408496626621557,"score_gpt":0.25571941673726994,"score_spread":0.2116344504710544,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378192263","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016815992,0.00014506398,0.97717834,0.00038820246,0.00005125898,0.000033073797,0.000033680124,0.0003372073,0.005017149],"genre_scores_gemma":[0.879606,0.000097270124,0.109790795,0.00015770171,0.00004084614,0.000143488,0.000061535,0.00005240998,0.010049957],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998661,0.000029370196,0.000005124493,0.00002880828,0.000039324947,0.000031206917],"domain_scores_gemma":[0.99977,0.00009736398,0.00003102185,0.000012651593,0.000059739872,0.000029303641],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00046753522,0.0005915361,0.00057991175,0.00020208473,0.0003060992,0.00043025982,0.0007673238,0.0010656401,0.0029645],"category_scores_gemma":[0.000839874,0.00030575006,0.0003349977,0.00015495621,0.000638319,0.000392118,0.0008555675,0.0011298723,0.00035273627],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000027513614,0.000013446021,0.00012602868,0.000023753757,0.000010006729,0.00005995152,0.000021970141,0.9800432,0.0012441551,0.0049903314,0.0005549063,0.012884671],"study_design_scores_gemma":[0.0000055529913,0.0000133800095,0.000019905703,0.0000018746324,0.000001385654,0.0000039379606,0.0000016695109,0.99860066,0.00010478195,0.0010762189,0.00016930015,0.0000012960122],"about_ca_topic_score_codex":0.0069202003,"about_ca_topic_score_gemma":0.0067866785,"teacher_disagreement_score":0.0069202003,"about_ca_system_score_codex":0.00074093125,"about_ca_system_score_gemma":0.0011624027,"threshold_uncertainty_score":0.013759792},"labels":[],"label_agreement":null},{"id":"W4378701136","doi":"10.1016/j.sysconle.2023.105563","title":"Approximated multi-agent fitted Q iteration","year":2023,"lang":"en","type":"article","venue":"Systems & Control Letters","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Group for Research in Decision Analysis","funders":"","keywords":"Computation; Reinforcement learning; Mathematical optimization; Computer science; Mathematics; Function (biology); Property (philosophy); Applied mathematics; Algorithm; Artificial intelligence","score_opus":0.021773525130426286,"score_gpt":0.24165134738091315,"score_spread":0.21987782225048685,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378701136","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014971646,0.00012287841,0.98087263,0.00017573006,0.00006648818,0.000050878498,0.000030513258,0.00019709139,0.003512224],"genre_scores_gemma":[0.7243273,0.00010565606,0.2660237,0.00018345192,0.000047768564,0.00028504213,0.00012524254,0.00013458698,0.008767319],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99921536,0.00029868374,0.000041221047,0.00013034708,0.0001704241,0.00014392514],"domain_scores_gemma":[0.99695396,0.0018341833,0.00018125403,0.00018548033,0.00066154386,0.0001836279],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021348158,0.0008899668,0.0020589973,0.00068729644,0.00073881954,0.0015573567,0.0018833822,0.0025799286,0.0050599673],"category_scores_gemma":[0.007973687,0.00075395755,0.0007750132,0.0005159727,0.0017333613,0.0011360229,0.0017255842,0.0015531013,0.0009712466],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000081658785,0.000028445927,0.00033838686,0.00004859065,0.000022043027,0.000056836565,0.000055325123,0.9780552,0.00036433156,0.009348861,0.0005272527,0.011073127],"study_design_scores_gemma":[0.000008326342,0.000009336584,0.000020792239,0.0000036619053,0.0000017643255,0.000004621292,0.000003690348,0.9983741,0.00006578264,0.001409488,0.000096349955,0.000001991988],"about_ca_topic_score_codex":0.011087646,"about_ca_topic_score_gemma":0.006880278,"teacher_disagreement_score":0.011087646,"about_ca_system_score_codex":0.0014615762,"about_ca_system_score_gemma":0.0025177246,"threshold_uncertainty_score":0.022046208},"labels":[],"label_agreement":null},{"id":"W4378768695","doi":"10.48550/arxiv.2305.17198","title":"A Model-Based Solution to the Offline Multi-Agent Reinforcement Learning Coordination Problem","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada","keywords":"Reinforcement learning; Computer science; Reinforcement; Artificial intelligence; Psychology; Social psychology","score_opus":0.13688357938669765,"score_gpt":0.2271378487904828,"score_spread":0.09025426940378514,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378768695","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005884011,0.00005764412,0.9912368,0.0002041618,0.000023245948,0.000040436124,0.000033869976,0.0002448566,0.002274937],"genre_scores_gemma":[0.7012683,0.00011571499,0.29306012,0.00019389816,0.00005516446,0.00029813522,0.00014713679,0.00014196083,0.0047194967],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994067,0.00020760366,0.000022867907,0.00016932422,0.00011667498,0.00007681937],"domain_scores_gemma":[0.9984107,0.0008842403,0.00020589645,0.00019983841,0.00015504692,0.00014415196],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001165254,0.0010706903,0.001399329,0.0003402872,0.00048615388,0.0008967252,0.0016057359,0.0014256631,0.003045018],"category_scores_gemma":[0.0036901138,0.00057630596,0.00057679915,0.00031233282,0.0012027371,0.000944059,0.0017162487,0.0021814767,0.0005792282],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000030926276,0.000032698095,0.00021054228,0.000040115498,0.00001527466,0.000041531915,0.000034124438,0.9750556,0.0005851353,0.009889832,0.0007552734,0.013309011],"study_design_scores_gemma":[0.000009378692,0.000018557135,0.000030366338,0.000003461219,0.0000022541085,0.000009474406,0.000005949606,0.99441105,0.00015180143,0.0050671417,0.00028815505,0.0000022947645],"about_ca_topic_score_codex":0.0036092831,"about_ca_topic_score_gemma":0.0033388329,"teacher_disagreement_score":0.0036092831,"about_ca_system_score_codex":0.00078848633,"about_ca_system_score_gemma":0.0019484876,"threshold_uncertainty_score":0.010186613},"labels":[],"label_agreement":null},{"id":"W4378771708","doi":"10.48550/arxiv.2305.18246","title":"Provable and Practical: Efficient Exploration in Reinforcement Learning via Langevin Monte Carlo","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alberta Machine Intelligence Institute; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Markov chain Monte Carlo; Computer science; Monte Carlo method; Markov decision process; Posterior probability; Mathematical optimization; Scalability; Gaussian; Regret; Artificial intelligence; Algorithm; Markov process; Mathematics; Machine learning; Bayesian probability; Physics","score_opus":0.12894877218383827,"score_gpt":0.22685229743409413,"score_spread":0.09790352525025586,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378771708","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009180003,0.0002579501,0.98667383,0.0004445423,0.00003966222,0.000047383393,0.000042480686,0.00046573728,0.0028484939],"genre_scores_gemma":[0.7254192,0.00040081496,0.2687368,0.00047072765,0.00010192654,0.00047699624,0.00018055119,0.00038433302,0.0038285202],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988386,0.0005289766,0.000043422973,0.00016455156,0.00029399182,0.00013049527],"domain_scores_gemma":[0.9939002,0.00484415,0.00030622684,0.0003687497,0.00032898175,0.00025164607],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027106751,0.0012778443,0.0017064794,0.00049414777,0.00069914927,0.0012061088,0.0019689004,0.0016511974,0.0038882075],"category_scores_gemma":[0.013363583,0.0008625507,0.0007604792,0.00059329835,0.0024914474,0.0018873608,0.0026538079,0.0031869235,0.00063717],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007136999,0.000035040754,0.00047779412,0.000064073894,0.000027570271,0.00006752389,0.000056616183,0.943332,0.0005021087,0.03933984,0.0011763803,0.014849655],"study_design_scores_gemma":[0.000010043284,0.0000076152633,0.000017373954,0.0000051516154,0.0000021105618,0.000005358237,0.0000020958637,0.9880355,0.00010507445,0.01164158,0.00016553044,0.0000025766562],"about_ca_topic_score_codex":0.0047910456,"about_ca_topic_score_gemma":0.0050243097,"teacher_disagreement_score":0.0047910456,"about_ca_system_score_codex":0.0016322791,"about_ca_system_score_gemma":0.0025641506,"threshold_uncertainty_score":0.014335573},"labels":[],"label_agreement":null},{"id":"W4379881949","doi":"10.1007/s13369-023-07934-2","title":"Reinforcement Learning DDPG–PPO Agent-Based Control System for Rotary Inverted Pendulum","year":2023,"lang":"en","type":"article","venue":"Arabian Journal for Science and Engineering","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Inverted pendulum; Control theory (sociology); PID controller; Reinforcement learning; Benchmark (surveying); Controller (irrigation); Linear-quadratic regulator; Pendulum; Computer science; Nonlinear system; Control engineering; Engineering; Artificial intelligence; Control (management); Physics; Temperature control","score_opus":0.019102979118655116,"score_gpt":0.23912099867232542,"score_spread":0.2200180195536703,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4379881949","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0915784,0.00083823106,0.88077617,0.00058241945,0.00053252355,0.0002505845,0.00008921975,0.0018172045,0.023535311],"genre_scores_gemma":[0.97033465,0.00013223053,0.024604473,0.00009903751,0.00004175582,0.00014948129,0.000039390794,0.000018259416,0.004580765],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997881,0.00003616526,0.000013744574,0.00006366989,0.000062580104,0.000035833957],"domain_scores_gemma":[0.9997774,0.0000464383,0.000033281012,0.000018049725,0.00010046114,0.000024443834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00037570033,0.00056447857,0.00074101915,0.00021464938,0.00058715447,0.0006414977,0.0010236835,0.0008073486,0.0029201312],"category_scores_gemma":[0.00067546713,0.00025676764,0.00027661773,0.00016219236,0.0004028468,0.00033557817,0.0008965053,0.0005357775,0.0005334284],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004896323,0.00021817167,0.0016285861,0.00034629917,0.00008694431,0.00056796335,0.00015449383,0.7866663,0.02720603,0.0062002465,0.0037380806,0.17269734],"study_design_scores_gemma":[0.000044151056,0.00016760832,0.00030349224,0.0000101588685,0.000014432002,0.000053619242,0.000008881433,0.9955225,0.0020475052,0.00059126015,0.0012281113,0.0000083496625],"about_ca_topic_score_codex":0.004682449,"about_ca_topic_score_gemma":0.0036981306,"teacher_disagreement_score":0.004682449,"about_ca_system_score_codex":0.0003229506,"about_ca_system_score_gemma":0.0007010905,"threshold_uncertainty_score":0.009768784},"labels":[],"label_agreement":null},{"id":"W4380319104","doi":"10.1007/978-3-319-08234-9_524-1","title":"Trustworthy Embodied Virtual Agents","year":2023,"lang":"en","type":"book-chapter","venue":"Encyclopedia of Computer Graphics and Games","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Embodied cognition; Trustworthiness; Computer science; Embodied agent; Human–computer interaction; Internet privacy; Artificial intelligence","score_opus":0.0178975753135585,"score_gpt":0.2319258496028856,"score_spread":0.21402827428932708,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4380319104","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023332706,0.0043498436,0.6738739,0.0012502735,0.00070560334,0.00008595599,0.000112876514,0.00093794626,0.2953509],"genre_scores_gemma":[0.5762234,0.004381285,0.1295442,0.00023259233,0.00020467052,0.0002128535,0.00024256851,0.00029228,0.2886662],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99960715,0.00008785616,0.000019947001,0.00006427082,0.00018799897,0.000032827196],"domain_scores_gemma":[0.99953544,0.00020378348,0.000044756114,0.00010473481,0.000066222565,0.00004498183],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003732152,0.00062405947,0.00035467095,0.00030415654,0.0005271648,0.0017704294,0.0007805958,0.0010073377,0.010723972],"category_scores_gemma":[0.0019736325,0.0003663934,0.0002189866,0.00023322945,0.0011558671,0.0018450972,0.0021222602,0.0011710632,0.002304705],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011328457,0.00005527223,0.00024995848,0.00025010668,0.000030000021,0.00028342757,0.0007030146,0.06286407,0.011095575,0.70429814,0.013380748,0.20667633],"study_design_scores_gemma":[0.00003698133,0.00013704885,0.0004274687,0.00020288299,0.000028567638,0.0004348804,0.0003157456,0.21805494,0.009449951,0.5710209,0.19984853,0.000042117605],"about_ca_topic_score_codex":0.00063816545,"about_ca_topic_score_gemma":0.00084793335,"teacher_disagreement_score":0.010723972,"about_ca_system_score_codex":0.00049165764,"about_ca_system_score_gemma":0.00046964214,"threshold_uncertainty_score":0.03587526},"labels":[],"label_agreement":null},{"id":"W4380684831","doi":"10.1109/tmlcn.2023.3285543","title":"Reinforcement Learning With Non-Cumulative Objective","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Machine Learning in Communications and Networking","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Computer science; Markov decision process; Bottleneck; Function (biology); Mathematical optimization; Process (computing); Artificial intelligence; Bellman equation; Convergence (economics); Markov process; Mathematics; Statistics","score_opus":0.028966426693772064,"score_gpt":0.27923764646791815,"score_spread":0.25027121977414607,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4380684831","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015583428,0.0002067775,0.98155606,0.00028579443,0.000045113582,0.00004774833,0.0000249508,0.00017587631,0.00207419],"genre_scores_gemma":[0.8054505,0.00037209457,0.18760361,0.00035228164,0.00009738117,0.00021418375,0.00009866326,0.00008580861,0.0057255113],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99851304,0.0006259601,0.00007730013,0.00032480515,0.0002843297,0.00017451796],"domain_scores_gemma":[0.99582386,0.0028741793,0.00035980408,0.00032090416,0.00043699518,0.00018423599],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031034506,0.0012221542,0.0012697309,0.00038000298,0.00035941508,0.0012280266,0.0013187445,0.001187026,0.001724313],"category_scores_gemma":[0.0084018465,0.00035030648,0.00049570575,0.00048893975,0.0017520231,0.002225818,0.0014156733,0.002220168,0.0002980849],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000103781385,0.00008447616,0.0007205085,0.0000973415,0.000052864787,0.000060812832,0.00008408064,0.81550056,0.0012114896,0.13838114,0.0012926726,0.042410355],"study_design_scores_gemma":[0.000016792494,0.000035425805,0.000060177936,0.0000066975226,0.0000068445183,0.000008186427,0.0000044239855,0.9711585,0.00036581798,0.027983427,0.00034682662,0.00000679496],"about_ca_topic_score_codex":0.0032247226,"about_ca_topic_score_gemma":0.0029277657,"teacher_disagreement_score":0.0032247226,"about_ca_system_score_codex":0.0015791962,"about_ca_system_score_gemma":0.0018386976,"threshold_uncertainty_score":0.016412795},"labels":[],"label_agreement":null},{"id":"W4381730562","doi":"10.1109/radarconf2351548.2023.10149670","title":"Priority-based Task Scheduling in Dynamic Environments for Cognitive MFR via Transfer DRL","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Defence Research and Development Canada; University of Toronto","funders":"","keywords":"Computer science; Reinforcement learning; Scheduling (production processes); Adaptability; Transfer of learning; Dynamic priority scheduling; Distributed computing; Radar; Task analysis; Real-time computing; Adaptation (eye); Task (project management); Artificial intelligence; Computer network; Engineering; Systems engineering","score_opus":0.01586660077822366,"score_gpt":0.2675579138119542,"score_spread":0.25169131303373055,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4381730562","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043984182,0.00013094777,0.9520797,0.00021317122,0.000039478164,0.00004813805,0.000023572455,0.00067944045,0.002801273],"genre_scores_gemma":[0.92632544,0.00005771234,0.071419545,0.00012769374,0.000018596762,0.00007545133,0.000031712454,0.000058573452,0.0018853338],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996625,0.0000733844,0.000018412986,0.00007836882,0.000095590796,0.000071679046],"domain_scores_gemma":[0.99913967,0.00036024032,0.00014063368,0.00012232679,0.00014885886,0.00008817231],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00091012416,0.0005030876,0.0004962802,0.00025923882,0.00029569905,0.000544433,0.0011554506,0.00065281213,0.0020323067],"category_scores_gemma":[0.0035440405,0.00027842118,0.00026650913,0.00022037279,0.0006751235,0.0011850033,0.0012053072,0.0012789761,0.00046940474],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013495955,0.00013131888,0.00086123124,0.000048295606,0.000014492232,0.0000827411,0.000100569174,0.8953771,0.008968251,0.00815787,0.000986742,0.08513644],"study_design_scores_gemma":[0.0000046695664,0.000021287311,0.000057647456,0.0000024414787,0.0000011862799,0.000007622963,0.0000048864395,0.9965321,0.00068095274,0.0025246183,0.00015987182,0.0000026669413],"about_ca_topic_score_codex":0.0031918306,"about_ca_topic_score_gemma":0.0032765144,"teacher_disagreement_score":0.0031918306,"about_ca_system_score_codex":0.00083524623,"about_ca_system_score_gemma":0.0011995061,"threshold_uncertainty_score":0.006798744},"labels":[],"label_agreement":null},{"id":"W4382050679","doi":"10.1109/icuas57906.2023.10156452","title":"Coordinated Multi-Robot Exploration using Reinforcement Learning","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Regina","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Robot; Robotics; Focus (optics); Process (computing); Autonomous agent; Human–computer interaction; Machine learning","score_opus":0.10639687791579537,"score_gpt":0.31741472475726124,"score_spread":0.21101784684146585,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4382050679","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09320889,0.00047709857,0.9025642,0.0002591218,0.000050810642,0.00008271034,0.000032563745,0.0006686386,0.0026558295],"genre_scores_gemma":[0.97101855,0.00007391402,0.028037513,0.00004411807,0.000013911789,0.00007043657,0.000023145541,0.000016374734,0.00070210756],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994375,0.00020554935,0.00002634901,0.000116685034,0.0001254417,0.000088455105],"domain_scores_gemma":[0.9985335,0.00076263654,0.0002688107,0.00013051105,0.00017442436,0.00013023378],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011856293,0.0007486288,0.0010286822,0.0003420811,0.00035645545,0.000586466,0.0011461021,0.00067704037,0.0008128091],"category_scores_gemma":[0.0023624934,0.0002900494,0.00040229975,0.00027056684,0.00093275454,0.00064604904,0.0009934419,0.0008386117,0.00014686113],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000054659682,0.000043235206,0.00043045174,0.000019784573,0.000026249976,0.00004251748,0.000021810985,0.986241,0.0008267389,0.001423295,0.00017650085,0.010693829],"study_design_scores_gemma":[0.000009897007,0.000026530508,0.000049059076,0.0000015890274,0.000002658799,0.0000057298107,0.0000028688794,0.9988506,0.00015666941,0.0008186609,0.000073921736,0.0000018282078],"about_ca_topic_score_codex":0.0038830505,"about_ca_topic_score_gemma":0.002642134,"teacher_disagreement_score":0.0038830505,"about_ca_system_score_codex":0.000726689,"about_ca_system_score_gemma":0.0010090908,"threshold_uncertainty_score":0.0077209473},"labels":[],"label_agreement":null},{"id":"W4382197597","doi":"10.21203/rs.3.rs-3080402/v1","title":"Long short-term prediction guides human metacognitive reinforcement learning","year":2023,"lang":"en","type":"preprint","venue":"Research Square","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Samsung; Ministry of Science and ICT, South Korea; University of Oxford; Korea Advanced Institute of Science and Technology; National Research Foundation; National Research Foundation of Korea; Somerville College, University of Oxford","keywords":"Term (time); Reinforcement learning; Metacognition; Reinforcement; Cognitive psychology; Psychology; Computer science; Artificial intelligence; Machine learning; Social psychology; Cognition; Neuroscience","score_opus":0.16443007410398042,"score_gpt":0.4243602209578892,"score_spread":0.2599301468539088,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4382197597","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.66853625,0.00051981426,0.29578385,0.0013868695,0.00037969794,0.00008762444,0.00034344292,0.0013148285,0.031647697],"genre_scores_gemma":[0.98534113,0.000074007105,0.011278818,0.00007491212,0.00001519371,0.000020221267,0.00010117752,0.00010553175,0.0029889983],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996741,0.000101073645,0.000014654376,0.0001245698,0.000051968647,0.000033615357],"domain_scores_gemma":[0.99643826,0.0020555202,0.00038314174,0.00046419507,0.0004260672,0.00023276683],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009715817,0.0003390132,0.00026766982,0.00017332585,0.00020809352,0.0016473416,0.0006206193,0.0008303801,0.00569427],"category_scores_gemma":[0.012847542,0.00037328497,0.0002156992,0.00015625625,0.00045145626,0.0014681462,0.00056464213,0.0012077211,0.0010637484],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023004836,0.0010572894,0.06639468,0.0003955744,0.00035798876,0.00042692776,0.0022235163,0.15014577,0.11894137,0.07447998,0.016409397,0.566867],"study_design_scores_gemma":[0.00013701855,0.0005535002,0.046548247,0.00008371496,0.00011006989,0.00017705072,0.00035428893,0.81541234,0.020536155,0.11010919,0.0058806255,0.00009789053],"about_ca_topic_score_codex":0.0026557634,"about_ca_topic_score_gemma":0.0024294124,"teacher_disagreement_score":0.00569427,"about_ca_system_score_codex":0.0004282354,"about_ca_system_score_gemma":0.000638686,"threshold_uncertainty_score":0.019049227},"labels":[],"label_agreement":null},{"id":"W4382202736","doi":"10.1609/aaai.v37i10.26371","title":"Learning to Shape Rewards Using a Game of Two Partners","year":2023,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs; University of Alberta; UK Research and Innovation; Alberta Machine Intelligence Institute; Compute Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Computer science; Task (project management); Function (biology); Convergence (economics); Construct (python library); Markov decision process; Artificial intelligence; State (computer science); Domain (mathematical analysis); Machine learning; Markov process; Algorithm; Mathematics; Engineering","score_opus":0.15084227496424973,"score_gpt":0.37279789706808497,"score_spread":0.22195562210383524,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4382202736","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.092864364,0.0001195796,0.8928725,0.0007025832,0.00004799591,0.0003020561,0.000057466204,0.00055365334,0.012479752],"genre_scores_gemma":[0.8495136,0.00008640532,0.14352928,0.00021103362,0.000019240491,0.00038067737,0.00005069852,0.000066775414,0.00614219],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989791,0.00048081172,0.00004521236,0.00018639583,0.00017659995,0.00013191049],"domain_scores_gemma":[0.9972474,0.0017552455,0.00027469202,0.00022732321,0.00016499285,0.00033027958],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017828947,0.0012486486,0.0008365549,0.00034218872,0.0007252915,0.001009763,0.0016016337,0.0017300649,0.0048851063],"category_scores_gemma":[0.006974917,0.0005172927,0.00068444916,0.0002411776,0.0019498272,0.0016775014,0.0021765716,0.0017616642,0.00055435044],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027136927,0.00022061073,0.0014562544,0.00009497629,0.000072385265,0.00034470414,0.00038129243,0.88352305,0.004889905,0.0694508,0.0012484343,0.038046125],"study_design_scores_gemma":[0.000047946844,0.000085206244,0.00010442823,0.00001021002,0.000009220837,0.000037155933,0.000027860211,0.9792367,0.0007736336,0.018644521,0.0010115207,0.0000116119045],"about_ca_topic_score_codex":0.0029411567,"about_ca_topic_score_gemma":0.0029876686,"teacher_disagreement_score":0.0048851063,"about_ca_system_score_codex":0.0010351903,"about_ca_system_score_gemma":0.0015709059,"threshold_uncertainty_score":0.016342282},"labels":[],"label_agreement":null},{"id":"W4382239700","doi":"10.1609/aaai.v37i6.25825","title":"PiCor: Multi-Task Deep Reinforcement Learning with Policy Correction","year":2023,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Task (project management); Computer science; Constraint (computer-aided design); Set (abstract data type); Artificial intelligence; Interference (communication); Machine learning; Channel (broadcasting); Mathematics; Engineering","score_opus":0.06260084640833517,"score_gpt":0.2989850664917558,"score_spread":0.23638422008342064,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4382239700","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009938279,0.00023902832,0.98372847,0.00019358874,0.00007603156,0.000116524425,0.000059375732,0.0037682531,0.0018804754],"genre_scores_gemma":[0.6186112,0.00020518208,0.3733476,0.00062327774,0.00008507055,0.00041750973,0.0002990174,0.00045710828,0.0059539266],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992281,0.00019023634,0.000038971008,0.0001879379,0.00022725659,0.0001274467],"domain_scores_gemma":[0.9989567,0.00039053013,0.00014488726,0.0002259854,0.0001668737,0.00011502459],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019264293,0.0012190447,0.0013468945,0.00034398554,0.000389271,0.00076838845,0.003089683,0.0014245014,0.0027491995],"category_scores_gemma":[0.0038477767,0.00066508155,0.00056737516,0.00033077525,0.0009797805,0.0013172759,0.0019327125,0.0027292701,0.0007552951],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025287102,0.00029248046,0.00088636193,0.00014307829,0.00010856945,0.00010716529,0.000053835032,0.80589086,0.0048122094,0.010408999,0.005087529,0.17195615],"study_design_scores_gemma":[0.000018288061,0.000039475966,0.000039800645,0.0000032429043,0.0000041047124,0.000009552774,0.0000015124575,0.99746144,0.00072498026,0.0012915626,0.00040126237,0.000004830844],"about_ca_topic_score_codex":0.005203695,"about_ca_topic_score_gemma":0.006875366,"teacher_disagreement_score":0.005203695,"about_ca_system_score_codex":0.0009635744,"about_ca_system_score_gemma":0.002451269,"threshold_uncertainty_score":0.01034683},"labels":[],"label_agreement":null},{"id":"W4382318178","doi":"10.1609/aaai.v37i8.26146","title":"Hypernetworks for Zero-Shot Transfer in Reinforcement Learning","year":2023,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Reinforcement learning; Computer science; Transfer of learning; Context (archaeology); Artificial intelligence; Task (project management); Set (abstract data type); Machine learning; Bellman equation; Mathematical optimization; Mathematics","score_opus":0.1140650363566056,"score_gpt":0.30957506420392306,"score_spread":0.19551002784731747,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4382318178","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018676808,0.0002566974,0.9774799,0.00026555324,0.00004794067,0.00006642163,0.00006974748,0.0007707645,0.0023661314],"genre_scores_gemma":[0.8593943,0.00019500482,0.13393341,0.00031570374,0.00005250466,0.00044746156,0.00024766193,0.00021965626,0.005194176],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999393,0.00025207337,0.000030490804,0.00015622184,0.00010031607,0.000067895315],"domain_scores_gemma":[0.9973998,0.001837236,0.00016728933,0.0002544436,0.00022618838,0.000115141425],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020385145,0.0010906268,0.0009612848,0.00060578604,0.00046875796,0.0008578492,0.0020938045,0.0016336371,0.004278556],"category_scores_gemma":[0.00820599,0.0006409981,0.00057507196,0.00041584703,0.001576023,0.0021864353,0.0019187743,0.0028337233,0.00060496357],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007849577,0.00005240587,0.00035155212,0.000048119186,0.00002790287,0.000056691497,0.00006242241,0.942253,0.0010475498,0.017404236,0.0009242505,0.037693422],"study_design_scores_gemma":[0.000005128734,0.000016085136,0.000023842536,0.0000044859466,0.0000020305736,0.0000050967674,0.0000033352496,0.98861825,0.00024967937,0.010895156,0.00017413031,0.0000028352474],"about_ca_topic_score_codex":0.0030298275,"about_ca_topic_score_gemma":0.0033187335,"teacher_disagreement_score":0.004278556,"about_ca_system_score_codex":0.0015638891,"about_ca_system_score_gemma":0.00090009416,"threshold_uncertainty_score":0.014313221},"labels":[],"label_agreement":null},{"id":"W4382334265","doi":"10.48550/arxiv.2306.14808","title":"Maximum State Entropy Exploration using Predecessor and Successor Representations","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Canada Excellence Research Chairs, Government of Canada; Canadian Institute for Advanced Research","keywords":"Successor cardinal; Computer science; Entropy (arrow of time); Exploratory research; Artificial intelligence; Machine learning; Mathematics; Sociology","score_opus":0.17693395941218948,"score_gpt":0.24389825658468828,"score_spread":0.0669642971724988,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4382334265","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17027529,0.00049374794,0.8238757,0.00060162146,0.000036674668,0.00008559494,0.0002241868,0.000741008,0.0036660966],"genre_scores_gemma":[0.9481664,0.00013718527,0.04955388,0.00010936122,0.000022930244,0.00014044628,0.00023078099,0.000056577643,0.0015824932],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99926287,0.0003085999,0.00004141542,0.00017174975,0.00012414956,0.00009117334],"domain_scores_gemma":[0.99590224,0.003006849,0.0003212828,0.00034785434,0.0002508951,0.00017092272],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016979818,0.0008246986,0.0012509443,0.00095436786,0.00045547503,0.0010278946,0.0013322784,0.001219859,0.0018570812],"category_scores_gemma":[0.0074822376,0.0006054086,0.0008155175,0.0006630174,0.0014992574,0.0024241933,0.0015215948,0.0016917089,0.00027072272],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016592987,0.000068729634,0.0018652094,0.000053352902,0.00003974482,0.000055039578,0.00008847323,0.94411296,0.00080191705,0.015747149,0.0006517895,0.0363497],"study_design_scores_gemma":[0.000009374507,0.00002615688,0.000100062185,0.0000063525463,0.0000039386837,0.0000074183495,0.000003830164,0.98771465,0.00021241345,0.01182674,0.00008465023,0.0000045369866],"about_ca_topic_score_codex":0.0028430736,"about_ca_topic_score_gemma":0.0035416032,"teacher_disagreement_score":0.0028430736,"about_ca_system_score_codex":0.0012947663,"about_ca_system_score_gemma":0.0014015643,"threshold_uncertainty_score":0.009394228},"labels":[],"label_agreement":null},{"id":"W4382653405","doi":"10.1002/9781119873747.ch2","title":"Markov Decision Process and Reinforcement Learning","year":2023,"lang":"en","type":"other","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Markov decision process; Reinforcement learning; Computer science; Q-learning; Markov process; Key (lock); Partially observable Markov decision process; Process (computing); Mathematical optimization; Markov chain; Bellman equation; Extension (predicate logic); Machine learning; Artificial intelligence; Markov model; Mathematics","score_opus":0.011509651708220893,"score_gpt":0.2672792631151527,"score_spread":0.25576961140693183,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4382653405","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0074842493,0.037339,0.79648083,0.0080966875,0.00079117407,0.000095449424,0.0004170909,0.00032548903,0.14896998],"genre_scores_gemma":[0.686605,0.062938645,0.18881454,0.0019090946,0.0019671277,0.00047342852,0.00065904483,0.00014176521,0.05649137],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9991534,0.0003648416,0.00004352664,0.00014116708,0.00022659278,0.00007064634],"domain_scores_gemma":[0.9986833,0.00095027965,0.000108048385,0.00006281003,0.00013008511,0.00006539088],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010589809,0.000825711,0.00075945456,0.0006922626,0.00040832252,0.0018969616,0.000787628,0.0015009315,0.0069797277],"category_scores_gemma":[0.003065538,0.00023142809,0.00050241617,0.0011449165,0.001878362,0.0016692502,0.0009919566,0.0023278377,0.0010967797],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000123136015,0.000023556804,0.00019400606,0.00013157475,0.00001743399,0.0000733198,0.000055136596,0.03699356,0.00014433141,0.9322,0.003371712,0.026782932],"study_design_scores_gemma":[0.000014778859,0.000026437201,0.0001938387,0.00011205842,0.000012153847,0.000074120566,0.000038128426,0.1003371,0.00017139452,0.85968673,0.039317958,0.000015301757],"about_ca_topic_score_codex":0.0046111215,"about_ca_topic_score_gemma":0.002642051,"teacher_disagreement_score":0.0069797277,"about_ca_system_score_codex":0.0018683802,"about_ca_system_score_gemma":0.0017579518,"threshold_uncertainty_score":0.023349524},"labels":[],"label_agreement":null},{"id":"W4383097435","doi":"10.1109/icra48891.2023.10160684","title":"Real-Time Reinforcement Learning for Vision-Based Robotics Utilizing Local and Remote Computers","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Computation; Workstation; Robotics; Robot; Artificial intelligence; Mobile robot; Resource (disambiguation); Distributed computing; Real-time computing; Operating system; Computer network","score_opus":0.02114241953396678,"score_gpt":0.27674038051919714,"score_spread":0.25559796098523035,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4383097435","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08488427,0.00027254078,0.9104891,0.00021636084,0.00007249946,0.000050756047,0.00000952826,0.0017117233,0.0022932],"genre_scores_gemma":[0.9311913,0.000075239615,0.06711752,0.00007960875,0.000017896711,0.0000589922,0.000016615395,0.000047576737,0.0013952383],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99962664,0.00008843573,0.000018633165,0.000113829774,0.00010108852,0.00005138721],"domain_scores_gemma":[0.99932563,0.00027714073,0.00008259297,0.00011999226,0.000110835084,0.00008392107],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007313928,0.0005010152,0.00053547596,0.00017596825,0.0002959275,0.0005074677,0.0010034499,0.0005021919,0.0015206207],"category_scores_gemma":[0.0018563573,0.0002593921,0.00025517595,0.00018681123,0.0008427512,0.0009238973,0.0009994883,0.0013573135,0.00035618438],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041181548,0.00026177207,0.0012774207,0.000091649956,0.000050771952,0.00013115161,0.00014800267,0.8109322,0.028212076,0.007837563,0.0011567916,0.14948878],"study_design_scores_gemma":[0.000019427209,0.00006739078,0.00013577048,0.0000031666054,0.0000054980906,0.000013511977,0.000007303973,0.99386686,0.0040466394,0.0014556505,0.00037447127,0.0000043417594],"about_ca_topic_score_codex":0.0020770393,"about_ca_topic_score_gemma":0.0020725825,"teacher_disagreement_score":0.0020770393,"about_ca_system_score_codex":0.0005976494,"about_ca_system_score_gemma":0.0008272483,"threshold_uncertainty_score":0.005087018},"labels":[],"label_agreement":null},{"id":"W4383097456","doi":"10.1109/icra48891.2023.10160572","title":"On Legible and Predictable Robot Navigation in Multi-Agent Environments","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Legibility; Predictability; Computer science; Robot; Generalization; Artificial intelligence; Human–computer interaction; Computer vision; Mathematics","score_opus":0.033752734947392074,"score_gpt":0.2678673256716823,"score_spread":0.23411459072429025,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4383097456","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026618907,0.0002290076,0.9701807,0.00023751285,0.000014581192,0.000030281353,0.000036264337,0.00024911485,0.0024036092],"genre_scores_gemma":[0.74872434,0.0005875831,0.24818575,0.000121858604,0.000046744524,0.00012990796,0.0001215144,0.000111016816,0.0019712131],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999092,0.00034841345,0.000053281994,0.00015361063,0.00028255372,0.00007025923],"domain_scores_gemma":[0.99569833,0.002757436,0.000581741,0.0005086458,0.00027461266,0.00017919029],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014058541,0.00070471654,0.00050857104,0.00059521897,0.0006025812,0.0011009759,0.0008658135,0.000883358,0.0013891126],"category_scores_gemma":[0.006549263,0.00041815342,0.00071055256,0.0004837499,0.0033429405,0.0025833948,0.0020138044,0.001246793,0.00022556641],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009144512,0.000043177453,0.0017685366,0.0001053297,0.000029345256,0.00020252803,0.0002476241,0.84363,0.0032221514,0.115174696,0.0003951369,0.03508999],"study_design_scores_gemma":[0.000012100813,0.00005801188,0.00042668026,0.000021732229,0.000009058934,0.000043530512,0.000027872895,0.8914736,0.0013483099,0.10563203,0.00093073444,0.00001631322],"about_ca_topic_score_codex":0.0033587373,"about_ca_topic_score_gemma":0.0032221272,"teacher_disagreement_score":0.0033587373,"about_ca_system_score_codex":0.0010279372,"about_ca_system_score_gemma":0.0010246187,"threshold_uncertainty_score":0.0074582696},"labels":[],"label_agreement":null},{"id":"W4383112356","doi":"10.1109/lra.2023.3292004","title":"SACHA: Soft Actor-Critic With Heuristic-Based Attention for Partially Observable Multi-Agent Path Finding","year":2023,"lang":"en","type":"article","venue":"IEEE Robotics and Automation Letters","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Heuristic; Path (computing); Observable; Computer science; Mathematical optimization; Artificial intelligence; Mathematics; Physics","score_opus":0.0428821579873367,"score_gpt":0.2687953178799828,"score_spread":0.2259131598926461,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4383112356","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0067976755,0.00022707795,0.989282,0.00020750193,0.00008242018,0.000059500722,0.00002978643,0.0008688942,0.0024451416],"genre_scores_gemma":[0.73724365,0.0002905347,0.25424224,0.0004617432,0.00012079746,0.0003857711,0.00017789991,0.00023330729,0.006844076],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99942374,0.00017312515,0.00002854832,0.00013284758,0.00016078635,0.00008093238],"domain_scores_gemma":[0.99842215,0.0009333038,0.00017851357,0.00010946859,0.00022432125,0.00013226247],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013321572,0.0014615699,0.0011727667,0.00044998346,0.0004544242,0.000850018,0.0020870077,0.0015540429,0.002704598],"category_scores_gemma":[0.0037670205,0.0006376496,0.00063154584,0.0003649112,0.0012208237,0.0008705578,0.001513963,0.0021906912,0.0006008203],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000052287724,0.000030368028,0.0003003409,0.000045656212,0.000042879736,0.00007389802,0.000038947313,0.9634426,0.0011613235,0.007930898,0.0014511886,0.02542959],"study_design_scores_gemma":[0.000007489999,0.0000086196615,0.000018044022,0.0000022043014,0.0000027534438,0.0000046011437,0.0000014389069,0.9981621,0.00015420283,0.0014211807,0.00021533706,0.0000020572622],"about_ca_topic_score_codex":0.0071543125,"about_ca_topic_score_gemma":0.007945433,"teacher_disagreement_score":0.0071543125,"about_ca_system_score_codex":0.0009468262,"about_ca_system_score_gemma":0.0021540874,"threshold_uncertainty_score":0.014225364},"labels":[],"label_agreement":null},{"id":"W4383424630","doi":"10.1007/978-3-031-36183-8_9","title":"Unified Emulation-Simulation Training Environment for Autonomous Cyber Agents","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Queen's University; Defence Research and Development Canada","funders":"","keywords":"Emulation; Computer science; Reinforcement learning; Fidelity; Process (computing); Artificial intelligence; Human–computer interaction; Intelligent agent; Simulation; Distributed computing; Operating system","score_opus":0.06123982952119357,"score_gpt":0.2813513022896291,"score_spread":0.22011147276843557,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4383424630","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0049676783,0.00004319881,0.98217946,0.000038899827,0.000026529806,0.00004268376,0.00008316765,0.0073324256,0.0052859127],"genre_scores_gemma":[0.26058185,0.00017907142,0.7218504,0.000097711054,0.000032135267,0.00057147106,0.0008183271,0.0017189906,0.014150068],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997775,0.000066080676,0.000012254089,0.00003905835,0.00007378077,0.000031369153],"domain_scores_gemma":[0.9997409,0.00009375961,0.000016267299,0.00006309929,0.000052555974,0.00003347409],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00048351276,0.00082087883,0.0007361879,0.0003964325,0.00043126877,0.00070243614,0.0020457122,0.00092494057,0.015636403],"category_scores_gemma":[0.00079379947,0.00042981262,0.0005620776,0.00025056012,0.00043777155,0.00089044264,0.0018070793,0.0010669271,0.0030567003],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037538056,0.0002595047,0.0006481839,0.00014223266,0.000042056647,0.00021552017,0.00018865889,0.77677524,0.016529983,0.053004205,0.009306974,0.14251204],"study_design_scores_gemma":[0.00002669366,0.000036378824,0.00007245302,0.000009532661,0.000006378344,0.000036141948,0.000007637069,0.9837388,0.0040768846,0.005250803,0.0067303344,0.000008080888],"about_ca_topic_score_codex":0.0015163544,"about_ca_topic_score_gemma":0.0018151306,"teacher_disagreement_score":0.015636403,"about_ca_system_score_codex":0.00035668354,"about_ca_system_score_gemma":0.00071607844,"threshold_uncertainty_score":0.052309036},"labels":[],"label_agreement":null},{"id":"W4383560595","doi":"10.54254/2755-2721/5/20230668","title":"Applications of deep reinforcement learning — Alphago","year":2023,"lang":"en","type":"article","venue":"Applied and Computational Engineering","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Reinforcement learning; Artificial intelligence; Deep learning; Field (mathematics); Computer science; Engineering","score_opus":0.006777116084209396,"score_gpt":0.21245335243257704,"score_spread":0.20567623634836765,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4383560595","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04408945,0.0051656333,0.9195495,0.0022114674,0.00035182154,0.00009151436,0.00008393426,0.001144718,0.027311953],"genre_scores_gemma":[0.9204552,0.0017666366,0.06936482,0.0005865912,0.00012434293,0.000083200604,0.000082141305,0.00007191436,0.007465166],"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996131,0.0001377913,0.000019962306,0.000071204784,0.00010795958,0.000049992825],"domain_scores_gemma":[0.9990989,0.0005397359,0.00007763125,0.00007039164,0.00014795605,0.0000653697],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008334296,0.0006112957,0.00057012687,0.00042924203,0.00031358656,0.00072980474,0.0008817288,0.00092731207,0.0028610197],"category_scores_gemma":[0.0031618073,0.00023533721,0.00031009567,0.0003355175,0.0008390572,0.00089653255,0.0013972484,0.0014345024,0.00037617664],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019668406,0.0002053152,0.002580043,0.000226947,0.00011313362,0.00016622049,0.00011954022,0.7108754,0.0033402748,0.062187046,0.0049193157,0.21507008],"study_design_scores_gemma":[0.000016901236,0.00007464547,0.00017247023,0.000028014698,0.0000114066925,0.00003917562,0.00001378396,0.9729926,0.00080183777,0.02253458,0.003306523,0.000008094259],"about_ca_topic_score_codex":0.0032509982,"about_ca_topic_score_gemma":0.0032735053,"teacher_disagreement_score":0.0032509982,"about_ca_system_score_codex":0.0006822174,"about_ca_system_score_gemma":0.00085760484,"threshold_uncertainty_score":0.009571075},"labels":[],"label_agreement":null},{"id":"W4384574988","doi":"10.23952/jano.5.2023.2.01","title":"Multi-step actor-critic framework for reinforcement learning in continuous control","year":2023,"lang":"en","type":"article","venue":"Journal of Applied and Numerical Optimization","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Reinforcement learning; Temporal difference learning; Computer science; Sequence (biology); State (computer science); Control (management); Artificial intelligence; Reinforcement; Optimal control; Action (physics); Machine learning; Mathematical optimization; Mathematics; Algorithm; Engineering","score_opus":0.015598831028528763,"score_gpt":0.2672656284476811,"score_spread":0.2516667974191523,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4384574988","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0030507613,0.00066689425,0.99311125,0.00021385754,0.000079565725,0.0000269366,0.000028151188,0.0002326914,0.0025898365],"genre_scores_gemma":[0.82188034,0.0010932004,0.1685811,0.00023267636,0.00015309795,0.00036046348,0.00012727416,0.000099018995,0.007472818],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99945134,0.00016912683,0.000030083667,0.00012724938,0.00016238264,0.000059810773],"domain_scores_gemma":[0.9993895,0.00031306586,0.00006114058,0.000037451664,0.00015479907,0.00004400738],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011183945,0.0012314921,0.0011702011,0.00036136314,0.0003628323,0.00090340414,0.0015009579,0.001165717,0.0025497116],"category_scores_gemma":[0.0018146867,0.00042954943,0.000686593,0.000401429,0.0010300115,0.0007632414,0.000780482,0.0021980978,0.0003372139],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000034726785,0.00002290324,0.00022473931,0.0000887922,0.00003231914,0.0000758811,0.000035844994,0.9522182,0.0008950193,0.029983897,0.00082104444,0.015566643],"study_design_scores_gemma":[0.000005521719,0.000009964481,0.000020885822,0.000003075352,0.0000033057895,0.0000045401625,0.0000013068985,0.9966523,0.00010322637,0.0028458731,0.00034757314,0.000002501803],"about_ca_topic_score_codex":0.009158794,"about_ca_topic_score_gemma":0.0065578627,"teacher_disagreement_score":0.009158794,"about_ca_system_score_codex":0.0011683772,"about_ca_system_score_gemma":0.0016926618,"threshold_uncertainty_score":0.018210948},"labels":[],"label_agreement":null},{"id":"W4385190059","doi":"10.1145/3583133.3590596","title":"Neuroevolution for Autonomous Cyber Defense","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Neuroevolution; Computer science; Reinforcement learning; Observability; Artificial intelligence; Task (project management); Domain (mathematical analysis); Adversary; Python (programming language); Artificial neural network; Evolutionary algorithm; Machine learning; Computer security; Systems engineering; Engineering","score_opus":0.029155177021413986,"score_gpt":0.2647024273516445,"score_spread":0.23554725033023052,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385190059","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014579886,0.008018708,0.92362976,0.0025484278,0.00044885746,0.00009985379,0.00018409113,0.0015325245,0.04895779],"genre_scores_gemma":[0.44604164,0.009157132,0.51919997,0.0007858274,0.00046054195,0.0005973284,0.0005571208,0.00056196016,0.022638494],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997918,0.00006463505,0.000010537132,0.000036400375,0.00008589833,0.000010810629],"domain_scores_gemma":[0.99972445,0.00014941405,0.000024532075,0.00004103059,0.00004054163,0.000020059018],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004589746,0.00045159593,0.00043567436,0.00044422282,0.00041925124,0.000995873,0.00064505264,0.0007996636,0.003948819],"category_scores_gemma":[0.0013487571,0.00021346807,0.00043194494,0.00042048865,0.0010285234,0.00070102705,0.0011047097,0.0018666231,0.0008595405],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000033959604,0.00003679993,0.00089081156,0.00019045261,0.000058533857,0.0000700278,0.0001319554,0.5679328,0.0034363142,0.26558796,0.007920276,0.15371014],"study_design_scores_gemma":[0.000014089984,0.00004856219,0.00044689164,0.00006485208,0.000011715771,0.00006832625,0.000025928204,0.80346555,0.0010271173,0.15437628,0.040429812,0.000020848565],"about_ca_topic_score_codex":0.0024476675,"about_ca_topic_score_gemma":0.002029145,"teacher_disagreement_score":0.003948819,"about_ca_system_score_codex":0.00099017,"about_ca_system_score_gemma":0.0006725867,"threshold_uncertainty_score":0.013210058},"labels":[],"label_agreement":null},{"id":"W4385299175","doi":"10.1109/tai.2023.3299252","title":"Facilitating Sim-to-Real by Intrinsic Stochasticity of Real-Time Simulation in Reinforcement Learning for Robot Manipulation","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia, Okanagan Campus; University of British Columbia; University of Victoria","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Computer science; Robot; Artificial intelligence; Heuristic; Robotics; Generalizability theory; Task (project management); Simulation; Engineering","score_opus":0.06739551366683552,"score_gpt":0.3288240012675382,"score_spread":0.2614284876007027,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385299175","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029296996,0.00009833808,0.9681777,0.00013332871,0.000026469004,0.000051171595,0.000013616217,0.0004532132,0.0017491857],"genre_scores_gemma":[0.8814804,0.00014236379,0.11734407,0.000121254634,0.000019591542,0.00016455878,0.00003822399,0.00011411129,0.0005754204],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985329,0.0008056133,0.000078667246,0.00016714925,0.00033965442,0.0000760148],"domain_scores_gemma":[0.99399734,0.0038629642,0.00080095243,0.0009074512,0.00028956265,0.00014173484],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023794079,0.00052060164,0.0004918299,0.0002696169,0.0002605474,0.00076823693,0.00088048587,0.00059297157,0.0014753515],"category_scores_gemma":[0.01007528,0.0003472243,0.00047706056,0.00022387439,0.0016367236,0.0013086705,0.0013484232,0.001383628,0.0002616155],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021346791,0.00012201851,0.0014574069,0.000116227406,0.000044027067,0.00008725602,0.00013015264,0.92082995,0.011828962,0.036216605,0.00028660806,0.028667258],"study_design_scores_gemma":[0.000015810621,0.00007387058,0.00014917502,0.000009677457,0.000006376127,0.000027437261,0.000008260975,0.9880784,0.0036907424,0.007197446,0.0007344907,0.000008332073],"about_ca_topic_score_codex":0.000772948,"about_ca_topic_score_gemma":0.00074485363,"teacher_disagreement_score":0.0023794079,"about_ca_system_score_codex":0.0005800649,"about_ca_system_score_gemma":0.0010203191,"threshold_uncertainty_score":0.012583613},"labels":[],"label_agreement":null},{"id":"W4385384974","doi":"10.1007/978-3-031-33242-5_9","title":"Hyperparameter Tuning for an Enhanced Self-Attention-Based Actor-Critical DDPG Framework","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes on data engineering and communications technologies","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Reinforcement learning; Computer science; Hyperparameter; Architecture; Artificial intelligence; Inverted pendulum; Machine learning","score_opus":0.06311031849023004,"score_gpt":0.3056693980039892,"score_spread":0.24255907951375916,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385384974","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0067330273,0.00022902292,0.989869,0.00015216664,0.000039367987,0.000036046913,0.000025361951,0.00042269094,0.0024933154],"genre_scores_gemma":[0.694894,0.00030152532,0.29391867,0.00034866115,0.00012295289,0.00032186386,0.00016282087,0.0004089715,0.009520555],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995795,0.0001612823,0.000020064108,0.00010175799,0.00007530728,0.00006215283],"domain_scores_gemma":[0.99932885,0.000419037,0.000036534595,0.00005140106,0.000108057415,0.00005612551],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012385418,0.0009544038,0.001124878,0.00040295496,0.00039913662,0.0010773127,0.0024187153,0.002031766,0.005759635],"category_scores_gemma":[0.0025140785,0.0005393418,0.0005073693,0.00029406478,0.0009140008,0.0011137341,0.0022731135,0.0018546846,0.00096309214],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001223189,0.000062556894,0.00016641416,0.00007917071,0.000038117927,0.000076033895,0.00007904012,0.9254186,0.002861459,0.017298857,0.0017499281,0.052047383],"study_design_scores_gemma":[0.0000065978593,0.0000092087885,0.00000888993,0.0000028023976,0.0000026531557,0.0000043136733,0.000002199235,0.9975169,0.00019706748,0.0020601852,0.00018748119,0.0000016678214],"about_ca_topic_score_codex":0.0038195834,"about_ca_topic_score_gemma":0.0038561693,"teacher_disagreement_score":0.005759635,"about_ca_system_score_codex":0.0008173803,"about_ca_system_score_gemma":0.0010694916,"threshold_uncertainty_score":0.019267857},"labels":[],"label_agreement":null},{"id":"W4385462630","doi":"10.1016/j.artint.2023.103989","title":"Learning reward machines: A study in partially observable reinforcement learning","year":2023,"lang":"en","type":"article","venue":"Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University; Vector Institute; University of Toronto","funders":"Fondo Nacional de Desarrollo Científico y Tecnológico; Agencia Nacional de Investigación y Desarrollo; Government of Ontario; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research; Microsoft Research","keywords":"Reinforcement learning; Observable; Computer science; Set (abstract data type); Artificial intelligence; Task (project management); Function (biology); Optimization problem; Representation (politics); Decomposition; Learning automata; Mathematical optimization; Automaton; Machine learning; Mathematics; Algorithm","score_opus":0.08641297792112246,"score_gpt":0.3260962182458153,"score_spread":0.23968324032469285,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385462630","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04996615,0.005683934,0.93184566,0.0023796314,0.0001197093,0.00006414346,0.00005469507,0.00015005577,0.009736048],"genre_scores_gemma":[0.89275366,0.004374453,0.09637969,0.00030853105,0.0003580289,0.0001726021,0.00007372163,0.000096193326,0.005483064],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976562,0.0014074647,0.00008953333,0.0002974244,0.0003752867,0.00017392363],"domain_scores_gemma":[0.9643662,0.032428436,0.001096032,0.00062674296,0.0010795923,0.00040299122],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042992574,0.001016908,0.0016562341,0.0007797962,0.000695892,0.0024944355,0.0020711068,0.002408694,0.0023318988],"category_scores_gemma":[0.03181319,0.0008302978,0.0009455231,0.001373834,0.0038596292,0.004742741,0.0012705426,0.0037922247,0.00018580754],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008375951,0.00010892735,0.0010362136,0.00026128444,0.00009352482,0.000078584744,0.00024381533,0.36413643,0.0004969015,0.60492766,0.00091980817,0.027613094],"study_design_scores_gemma":[0.000026009035,0.00005024113,0.00017325828,0.00002321245,0.000015145969,0.000018821476,0.000019843377,0.7905179,0.00017661988,0.20827688,0.0006907945,0.000011240812],"about_ca_topic_score_codex":0.005157377,"about_ca_topic_score_gemma":0.0023258177,"teacher_disagreement_score":0.005157377,"about_ca_system_score_codex":0.002024444,"about_ca_system_score_gemma":0.0017183446,"threshold_uncertainty_score":0.022736907},"labels":[],"label_agreement":null},{"id":"W4385473744","doi":"10.48550/arxiv.2307.16062","title":"Using Implicit Behavior Cloning and Dynamic Movement Primitive to Facilitate Reinforcement Learning for Robot Motion Planning","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Leverage (statistics); Computer science; Generalizability theory; Robot; Artificial intelligence; Heuristic; Motion planning; Motion (physics); Kinematics; Simulation; Mathematics","score_opus":0.212502004908271,"score_gpt":0.26361457496936935,"score_spread":0.051112570061098345,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385473744","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.053246863,0.00023957469,0.94166064,0.00023823681,0.000043952863,0.000121363526,0.0000958162,0.0025747176,0.0017787869],"genre_scores_gemma":[0.69580126,0.00013665987,0.3016371,0.00012674846,0.000017784965,0.00019822152,0.00022314611,0.0001289499,0.0017302021],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99953747,0.0001868484,0.000024499574,0.00013577445,0.00008000973,0.00003540477],"domain_scores_gemma":[0.998539,0.0007079616,0.00018028857,0.00036123986,0.00012702896,0.00008448908],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010752461,0.0007048932,0.0004903584,0.00028510665,0.00026187603,0.00037484049,0.0013374921,0.00065841724,0.0017535966],"category_scores_gemma":[0.003579765,0.0003573038,0.0004214405,0.0002732987,0.0010694762,0.0009667943,0.00087971566,0.0016621155,0.00043389614],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002803203,0.00039458481,0.0037310014,0.00025474012,0.00006670921,0.00014153728,0.00017495567,0.67162836,0.026010655,0.0113765625,0.0023369705,0.2836036],"study_design_scores_gemma":[0.000024229112,0.00008249065,0.0003333914,0.0000062877816,0.000006751526,0.000020955975,0.0000067092983,0.99200773,0.003672704,0.0029074738,0.00092417,0.0000070708306],"about_ca_topic_score_codex":0.003762633,"about_ca_topic_score_gemma":0.0050888727,"teacher_disagreement_score":0.003762633,"about_ca_system_score_codex":0.00056035543,"about_ca_system_score_gemma":0.0011357627,"threshold_uncertainty_score":0.007481456},"labels":[],"label_agreement":null},{"id":"W4385484633","doi":"10.1109/ijcnn54540.2023.10191867","title":"Reducing the Cost of Cycle-Time Tuning for Real-World Policy Optimization","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Alberta Machine Intelligence Institute","keywords":"Benchmark (surveying); Baseline (sea); Task (project management); Computer science; Robotics; Artificial intelligence; Reinforcement learning; Machine learning; Robot; Engineering","score_opus":0.028349866026891192,"score_gpt":0.30427996305451016,"score_spread":0.27593009702761895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385484633","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09756731,0.0034597246,0.8784577,0.0010959979,0.00040730718,0.0003767166,0.00012588508,0.007903416,0.0106059015],"genre_scores_gemma":[0.8860277,0.00038623952,0.110021904,0.00064024545,0.00005130507,0.00031908142,0.00016652419,0.0005583069,0.0018287883],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99806553,0.00068812864,0.0001400567,0.00041193332,0.00047846136,0.00021590843],"domain_scores_gemma":[0.99096453,0.00602927,0.00051937875,0.0014337512,0.0007307021,0.00032236043],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031273137,0.001449119,0.0015218208,0.0006713869,0.0006428295,0.0012856348,0.0020652371,0.001779411,0.0054637943],"category_scores_gemma":[0.022532614,0.00090891006,0.0006233812,0.00056919944,0.0012059469,0.0025507524,0.0015989618,0.0039792317,0.0013593924],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006686822,0.00064853404,0.003517731,0.00043355013,0.00014928893,0.00014372963,0.0003021658,0.6807085,0.009494872,0.011210448,0.0058603697,0.28686208],"study_design_scores_gemma":[0.00009543704,0.00017347679,0.00074576115,0.000048772832,0.00003291413,0.00006250956,0.000060108665,0.9833583,0.0032111534,0.009358403,0.002819778,0.000033263575],"about_ca_topic_score_codex":0.006122227,"about_ca_topic_score_gemma":0.00702154,"teacher_disagreement_score":0.006122227,"about_ca_system_score_codex":0.0012415928,"about_ca_system_score_gemma":0.0028809907,"threshold_uncertainty_score":0.018278241},"labels":[],"label_agreement":null},{"id":"W4385729101","doi":"10.1016/j.ins.2023.119481","title":"Reward shaping using convolutional neural network","year":2023,"lang":"en","type":"article","venue":"Information Sciences","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Markov decision process; Reinforcement learning; Convolutional neural network; Stochastic matrix; Representation (politics); Convolution (computer science); Artificial intelligence; Matrix (chemical analysis); Function (biology); Markov chain; Bellman equation; Algorithm; Artificial neural network; Markov process; Machine learning; Mathematical optimization; Mathematics; Statistics","score_opus":0.12628818183703894,"score_gpt":0.3253652114570306,"score_spread":0.19907702961999169,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385729101","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038558807,0.0007560504,0.95129025,0.00035157526,0.000120023455,0.00006574291,0.00012522098,0.0025194527,0.0062128557],"genre_scores_gemma":[0.8979196,0.00033611228,0.09607769,0.0002137071,0.000027593382,0.00009760748,0.00017838807,0.000104931365,0.005044412],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99974674,0.0000405974,0.000012601423,0.00007432341,0.00007241135,0.00005322288],"domain_scores_gemma":[0.9995976,0.00016816061,0.000059770315,0.000052460815,0.0000868033,0.000035258836],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005180545,0.0007198931,0.00057437434,0.00036901896,0.0002615489,0.000588022,0.0013540632,0.00076209847,0.0023008764],"category_scores_gemma":[0.0016795292,0.00032568848,0.0004322351,0.0003216125,0.000605925,0.00090666336,0.000808834,0.000998216,0.00038953387],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000089218556,0.00006377732,0.0007856735,0.00007094155,0.00003991745,0.00007204539,0.00003241115,0.86970335,0.0052410313,0.011472968,0.0015918139,0.110836826],"study_design_scores_gemma":[0.0000031118873,0.00001598111,0.00006416258,0.0000037189577,0.000003897248,0.000007776733,0.0000014408383,0.99544,0.00095340016,0.0031158477,0.00038740292,0.0000032171974],"about_ca_topic_score_codex":0.0070768204,"about_ca_topic_score_gemma":0.007491525,"teacher_disagreement_score":0.0070768204,"about_ca_system_score_codex":0.0011953043,"about_ca_system_score_gemma":0.0010602272,"threshold_uncertainty_score":0.014071286},"labels":[],"label_agreement":null},{"id":"W4385763897","doi":"10.24963/ijcai.2023/776","title":"Multi-Agent Advisor Q-Learning (Extended Abstract)","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Vector Institute; University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; University of Waterloo; Mitacs; University of Alberta; Alberta Machine Intelligence Institute; Compute Canada; Government of Canada; Vector Institute; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Computer science; Variety (cybernetics); Sample complexity; Artificial intelligence; Heuristic; Action (physics); Convergence (economics); Software deployment; Q-learning; Point (geometry); Operations research; Machine learning; Software engineering; Mathematics","score_opus":0.03721342908507978,"score_gpt":0.288069363828437,"score_spread":0.2508559347433572,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385763897","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0070333313,0.0003828689,0.9863657,0.00050475425,0.00013064702,0.00011987541,0.000111568144,0.0010759576,0.004275327],"genre_scores_gemma":[0.5350832,0.00041889068,0.44948408,0.0008417617,0.0002842519,0.0005284136,0.0005562437,0.00029412008,0.01250907],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99916697,0.00034518083,0.00003763406,0.00018288044,0.00016400333,0.00010338216],"domain_scores_gemma":[0.9971572,0.0017051068,0.00018605943,0.0003051321,0.00047032046,0.0001761713],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002200889,0.0010242922,0.0010453122,0.00040612064,0.00042454552,0.0008765646,0.0017266531,0.0014650103,0.015786177],"category_scores_gemma":[0.007990451,0.00032768954,0.00048694093,0.00065552664,0.0007979852,0.0010277521,0.001574631,0.0019043285,0.002090746],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027501245,0.00022310791,0.0013231717,0.00022558418,0.000058381404,0.00014366789,0.00009065584,0.72771126,0.0008822347,0.032373887,0.011833436,0.22485967],"study_design_scores_gemma":[0.000036710044,0.000038881582,0.00007803116,0.000011525821,0.0000044799203,0.000018760924,0.0000046552937,0.9881116,0.0003550309,0.0095619,0.001773756,0.000004768723],"about_ca_topic_score_codex":0.0044661807,"about_ca_topic_score_gemma":0.004927696,"teacher_disagreement_score":0.015786177,"about_ca_system_score_codex":0.0007746654,"about_ca_system_score_gemma":0.0014747032,"threshold_uncertainty_score":0.052810013},"labels":[],"label_agreement":null},{"id":"W4385764251","doi":"10.24963/ijcai.2023/783","title":"Rethinking Formal Models of Partially Observable Multiagent Decision Making (Extended Abstract)","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Grantová Agentura České Republiky","keywords":"Rotation formalisms in three dimensions; Computer science; Observable; Theoretical computer science; Regret; Mathematical economics; Artificial intelligence; Mathematics; Machine learning","score_opus":0.08508064556629286,"score_gpt":0.30600609572622145,"score_spread":0.2209254501599286,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385764251","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008240987,0.0002700348,0.984218,0.001090331,0.00006981506,0.00004381282,0.00010499689,0.00017458451,0.005787419],"genre_scores_gemma":[0.5187993,0.0008519674,0.47243038,0.0007090005,0.00023258929,0.0003426969,0.0002370755,0.00014409564,0.0062527675],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99725914,0.001583148,0.00014069669,0.00038795604,0.00043448497,0.00019463454],"domain_scores_gemma":[0.993458,0.0046449644,0.00051006663,0.00076329504,0.0004259287,0.00019778078],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003985195,0.00086808397,0.00055060344,0.00065689103,0.0006160505,0.002214755,0.0017715767,0.0013949941,0.004099291],"category_scores_gemma":[0.009998134,0.00050348986,0.001748504,0.0009961933,0.004146797,0.0037914144,0.002802603,0.0036229934,0.0005794399],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000013587678,0.000028592753,0.00020693608,0.000043200173,0.000015892485,0.00007654782,0.00019232632,0.066157594,0.0003179531,0.92579603,0.0005774174,0.006573904],"study_design_scores_gemma":[0.000011398938,0.000019519211,0.00008830836,0.000033297,0.000009961312,0.000027477952,0.00002596021,0.22390075,0.00027591878,0.7720213,0.0035759655,0.000010199805],"about_ca_topic_score_codex":0.0038001665,"about_ca_topic_score_gemma":0.0044168644,"teacher_disagreement_score":0.004099291,"about_ca_system_score_codex":0.0022785086,"about_ca_system_score_gemma":0.0015704713,"threshold_uncertainty_score":0.021075964},"labels":[],"label_agreement":null},{"id":"W4385767620","doi":"10.24963/ijcai.2023/31","title":"Towards a Better Understanding of Learning with Multiagent Teams","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Knowledge management; Computer science; Team learning; Population; Work (physics); Artificial intelligence; Cooperative learning; Engineering; Psychology; Mathematics education","score_opus":0.04272918891291047,"score_gpt":0.26190074510621725,"score_spread":0.21917155619330678,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385767620","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017230632,0.009051206,0.9355162,0.0128366435,0.0002985517,0.00007183228,0.00006924348,0.00012965036,0.024796026],"genre_scores_gemma":[0.7437492,0.009162887,0.23509613,0.0024180724,0.0018684793,0.00049205456,0.00014825759,0.00012961456,0.0069353045],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99724305,0.0017595853,0.000085403706,0.00032787703,0.000407841,0.00017627355],"domain_scores_gemma":[0.9933344,0.0047794245,0.0004577489,0.0006048169,0.00043676706,0.00038679238],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0048546293,0.0009779041,0.0013513506,0.0011932219,0.0009181175,0.0033862104,0.0030667828,0.0031910206,0.0042233174],"category_scores_gemma":[0.009591317,0.00060406193,0.0011547337,0.0011104272,0.0043744068,0.009021442,0.0024713804,0.005808441,0.00064062705],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000019534407,0.00006777265,0.00054270343,0.00016747501,0.000031913638,0.00006132941,0.00052365573,0.079130344,0.00025097452,0.90554595,0.001515382,0.012142917],"study_design_scores_gemma":[0.000012542629,0.000024763904,0.00018408264,0.000053082324,0.000005167395,0.00001782055,0.000084056264,0.15265894,0.000076835044,0.8427654,0.0041093417,0.000008069151],"about_ca_topic_score_codex":0.0022329043,"about_ca_topic_score_gemma":0.0010945768,"teacher_disagreement_score":0.0048546293,"about_ca_system_score_codex":0.002339399,"about_ca_system_score_gemma":0.0011385173,"threshold_uncertainty_score":0.025674045},"labels":[],"label_agreement":null},{"id":"W4385791631","doi":"10.1017/9781009258227.004","title":"Agent Architectures and Hierarchical Control","year":2023,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Generative grammar; Code (set theory); Implementation; Code generation; Artificial intelligence; Software engineering; Data science; Programming language; Key (lock)","score_opus":0.020991266457942914,"score_gpt":0.1998834696065149,"score_spread":0.178892203148572,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385791631","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032942963,0.048202667,0.22851706,0.009756843,0.0043781777,0.00009318421,0.0006806754,0.001701084,0.703376],"genre_scores_gemma":[0.1022663,0.03269129,0.078760274,0.0026416394,0.0016601024,0.00029603063,0.0010679165,0.00060996873,0.7800065],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9997646,0.00003889986,0.000011189394,0.000055292487,0.00010502802,0.000025007008],"domain_scores_gemma":[0.99984324,0.000052441537,0.00001293468,0.000030145446,0.000043019158,0.000018157356],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00029777954,0.0007090826,0.00045064086,0.000343634,0.0004923236,0.0024491502,0.0008020175,0.0010899019,0.03337338],"category_scores_gemma":[0.00082032324,0.00041237433,0.0004124441,0.0005276554,0.0012400569,0.001995457,0.0009038691,0.0020786182,0.008301057],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000019193285,0.000019206358,0.0001275362,0.00022277376,0.000023584875,0.00006741973,0.00022870966,0.0117899645,0.0008146363,0.7043871,0.1266017,0.15569817],"study_design_scores_gemma":[0.000009622871,0.00001871119,0.00021172398,0.0001679771,0.000008507418,0.0000966065,0.000063324434,0.010529673,0.00033378083,0.2674535,0.72109497,0.000011614429],"about_ca_topic_score_codex":0.0027458847,"about_ca_topic_score_gemma":0.0032031822,"teacher_disagreement_score":0.03337338,"about_ca_system_score_codex":0.0012650507,"about_ca_system_score_gemma":0.0011250988,"threshold_uncertainty_score":0.1116451},"labels":[],"label_agreement":null},{"id":"W4385791640","doi":"10.1017/9781009258227.019","title":"Multiagent Systems","year":2023,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Generative grammar; Implementation; Code (set theory); Artificial intelligence; Data science; Software engineering; Programming language","score_opus":0.0399365295312654,"score_gpt":0.2061894574563037,"score_spread":0.16625292792503832,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385791640","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0027050918,0.056415364,0.18287933,0.01424554,0.009889343,0.00020470764,0.0019390101,0.0024894914,0.7292321],"genre_scores_gemma":[0.046165705,0.027029157,0.06496801,0.002790315,0.0022951958,0.00038044486,0.0022924065,0.0004890237,0.8535898],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99945587,0.00011958884,0.000042637297,0.00012531894,0.00021772257,0.000038905837],"domain_scores_gemma":[0.9996069,0.0001339521,0.000027680198,0.000078392564,0.00011024675,0.000042890046],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005754385,0.0010268746,0.000776303,0.00061483984,0.00083391427,0.0032457428,0.0013354956,0.001792696,0.06636966],"category_scores_gemma":[0.0013902686,0.00045437028,0.0005622915,0.00081150513,0.0008548944,0.0023010867,0.0018852063,0.0023162938,0.024738614],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000028154367,0.0000484859,0.00028733438,0.0004751902,0.000053523883,0.0002351438,0.00032647976,0.009120316,0.0010958739,0.28715813,0.4215925,0.2795789],"study_design_scores_gemma":[0.000007886409,0.000019282741,0.00019117189,0.00015610349,0.000009481026,0.00018476679,0.00007102975,0.0052588624,0.00020663199,0.07562787,0.91825426,0.0000127489075],"about_ca_topic_score_codex":0.0017831093,"about_ca_topic_score_gemma":0.0025662356,"teacher_disagreement_score":0.06636966,"about_ca_system_score_codex":0.0010920402,"about_ca_system_score_gemma":0.0010360546,"threshold_uncertainty_score":0.22202861},"labels":[],"label_agreement":null},{"id":"W4385791672","doi":"10.1017/9781009258227.018","title":"Reinforcement Learning","year":2023,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Generative grammar; Code (set theory); Reinforcement learning; Implementation; Artificial intelligence; Software engineering; Programming language","score_opus":0.031001316819717507,"score_gpt":0.2074911771978943,"score_spread":0.1764898603781768,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385791672","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002824583,0.030167414,0.37303913,0.008333347,0.006948766,0.000157009,0.0029962806,0.005996612,0.5695369],"genre_scores_gemma":[0.041704413,0.016075686,0.10281852,0.0024818745,0.001160169,0.00026565487,0.0038655894,0.0012164579,0.8304116],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9996032,0.000060412087,0.000023579125,0.00010133333,0.00018444116,0.00002705364],"domain_scores_gemma":[0.99960274,0.00013453803,0.000020179445,0.0000606396,0.0001516896,0.000030224715],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00050876866,0.00093891675,0.00070068275,0.00043451763,0.00037980813,0.001744986,0.0012865859,0.0010951338,0.097589396],"category_scores_gemma":[0.0018463096,0.00035975687,0.0005121444,0.00061768584,0.00054951035,0.0017170202,0.0009164346,0.0021150773,0.042169295],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004487161,0.000053061303,0.00020666547,0.00034717156,0.000030093042,0.00007471362,0.000084645966,0.016156739,0.0009823764,0.09406121,0.34461236,0.543346],"study_design_scores_gemma":[0.000018437231,0.000045217923,0.0002973956,0.00026162702,0.00001574776,0.00020540236,0.00004199315,0.01919411,0.00083165395,0.06319319,0.9158662,0.000028997954],"about_ca_topic_score_codex":0.002162649,"about_ca_topic_score_gemma":0.0033296503,"teacher_disagreement_score":0.097589396,"about_ca_system_score_codex":0.0011334772,"about_ca_system_score_gemma":0.001028889,"threshold_uncertainty_score":0.32646906},"labels":[],"label_agreement":null},{"id":"W4385791676","doi":"10.1017/9781009258227.005","title":"Reasoning and Planning with Certainty","year":2023,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Generative grammar; Code (set theory); Implementation; Artificial intelligence; Space (punctuation); Data science; Software engineering; Programming language","score_opus":0.026440709450536913,"score_gpt":0.20447680039054347,"score_spread":0.17803609094000655,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385791676","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002645235,0.024224097,0.33256963,0.020775283,0.002331087,0.0001290099,0.001364354,0.0011284894,0.61483276],"genre_scores_gemma":[0.20113799,0.035288565,0.27381483,0.005994259,0.0024712586,0.0007240634,0.0030857606,0.0011043533,0.47637898],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9988431,0.00035259366,0.00008312014,0.00027781728,0.0003613161,0.00008208903],"domain_scores_gemma":[0.99894124,0.0006124201,0.00005908142,0.00018252658,0.00016547213,0.00003922272],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013734499,0.0009652812,0.0005973641,0.0005914501,0.0010069102,0.004881242,0.0012723046,0.0016923154,0.030240266],"category_scores_gemma":[0.0038453175,0.0006612347,0.0010773424,0.0007467727,0.0043098377,0.0055696736,0.0020017351,0.0038779427,0.007660903],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001213488,0.0000063583707,0.000056111756,0.00014042456,0.00001172512,0.000047441747,0.0002811939,0.003973412,0.00016448261,0.92040277,0.031864148,0.04303975],"study_design_scores_gemma":[0.000009674733,0.000010693094,0.000083244115,0.00021121213,0.0000071721465,0.000067869136,0.000122058475,0.0038232193,0.00025440435,0.7044479,0.290949,0.00001362312],"about_ca_topic_score_codex":0.0043630116,"about_ca_topic_score_gemma":0.0043781945,"teacher_disagreement_score":0.030240266,"about_ca_system_score_codex":0.0025835568,"about_ca_system_score_gemma":0.0025945073,"threshold_uncertainty_score":0.101163745},"labels":[],"label_agreement":null},{"id":"W4385791677","doi":"10.1017/9781009258227.009","title":"Deterministic Planning","year":2023,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Implementation; Generative grammar; Code (set theory); Artificial intelligence; Data science; Software engineering; Programming language","score_opus":0.04680683511438593,"score_gpt":0.2268041648941649,"score_spread":0.17999732977977898,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385791677","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0014685144,0.012235761,0.2792802,0.0050136745,0.0034865662,0.00014571504,0.004216238,0.0032108035,0.6909426],"genre_scores_gemma":[0.037996355,0.0107502015,0.12483597,0.0014101276,0.00062438357,0.00031896686,0.0049884585,0.0014365732,0.8176389],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9995633,0.00006332981,0.000029403387,0.00012834118,0.00018084046,0.000034732595],"domain_scores_gemma":[0.9996209,0.00013163144,0.000022044444,0.00007595856,0.00012046728,0.00002899758],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00044833228,0.0011012234,0.0006523485,0.0004872353,0.0006070345,0.0024198869,0.0014206946,0.0011297883,0.1265865],"category_scores_gemma":[0.00185283,0.0005873199,0.0007125611,0.00079336466,0.0008255937,0.0023176165,0.0012980995,0.001843655,0.04260095],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000505554,0.000030552535,0.00015574039,0.0004371875,0.000026054657,0.0001066606,0.00013455584,0.024855774,0.0007772081,0.42699844,0.2762811,0.27014616],"study_design_scores_gemma":[0.000011447851,0.000015375956,0.000112541566,0.0001897992,0.000010035188,0.00010194391,0.000038575494,0.010735403,0.00049297936,0.10930521,0.8789692,0.000017434128],"about_ca_topic_score_codex":0.0041889576,"about_ca_topic_score_gemma":0.006869976,"teacher_disagreement_score":0.1265865,"about_ca_system_score_codex":0.0017575058,"about_ca_system_score_gemma":0.0021015431,"threshold_uncertainty_score":0.423474},"labels":[],"label_agreement":null},{"id":"W4385890042","doi":"10.48550/arxiv.2308.07591","title":"Approximations and Learning for Continuous State and Action MDPs under Average Cost Criteria","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Mathematics; Markov decision process; Discretization; Ergodicity; Mathematical optimization; Applied mathematics; Convergence (economics); Markov process; Computer science; Mathematical analysis","score_opus":0.13854735717401917,"score_gpt":0.24295794452381714,"score_spread":0.10441058734979797,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385890042","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010974136,0.00027088312,0.9873739,0.00025028593,0.000024213807,0.000028292048,0.000035818328,0.000077162964,0.0009652829],"genre_scores_gemma":[0.74710953,0.000638656,0.24878013,0.00023629735,0.00009658821,0.0002910289,0.0002529662,0.00012582494,0.0024687715],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9971625,0.0011961949,0.00018117027,0.00048160335,0.00077906874,0.00019940668],"domain_scores_gemma":[0.9857244,0.011395558,0.0008991368,0.0007423021,0.0008769692,0.0003616927],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005995676,0.0011218815,0.001757879,0.0008580487,0.000572935,0.0019326977,0.0019677437,0.0018809779,0.0015932831],"category_scores_gemma":[0.025428182,0.0006389107,0.0010584525,0.0009165472,0.0029348386,0.0034874622,0.002598516,0.0031143585,0.00024217601],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000058073485,0.000027117656,0.00051841495,0.0000970169,0.00003421591,0.000040530926,0.00008862904,0.8855759,0.00040883245,0.10355548,0.00025672486,0.009339091],"study_design_scores_gemma":[0.000005295791,0.000015170391,0.000037365324,0.00000966666,0.0000029746188,0.0000073696,0.000006570261,0.97456247,0.00013808456,0.025087573,0.00012395292,0.0000034079887],"about_ca_topic_score_codex":0.0053754915,"about_ca_topic_score_gemma":0.002440563,"teacher_disagreement_score":0.005995676,"about_ca_system_score_codex":0.0030466602,"about_ca_system_score_gemma":0.0020661582,"threshold_uncertainty_score":0.03170854},"labels":[],"label_agreement":null},{"id":"W4386001266","doi":"10.1007/s10514-023-10127-3","title":"Learning scalable and efficient communication policies for multi-robot collision avoidance","year":2023,"lang":"en","type":"article","venue":"Autonomous Robots","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Office of Naval Research; Instituto de Ciencias del Mar y Limnología, Universidad Nacional Autónoma de México; Office of Naval Research Global; Canadian Institute for Advanced Research","keywords":"Computer science; Robot; Collision avoidance; Scalability; Robustness (evolution); Collision; Distributed computing; Broadcasting (networking); Artificial intelligence; Generalization; Reinforcement learning; Human–computer interaction; Computer security","score_opus":0.039689917230473715,"score_gpt":0.29919854479417823,"score_spread":0.2595086275637045,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386001266","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15599889,0.00016121096,0.8411382,0.00024941118,0.00003459092,0.00006583092,0.000022103592,0.0006726685,0.0016571499],"genre_scores_gemma":[0.96348345,0.000030631596,0.035695087,0.000039448994,0.000009863668,0.000048985672,0.000017933951,0.000021282882,0.0006533426],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99964607,0.00011222023,0.000018934761,0.000073936666,0.000087800705,0.000060984636],"domain_scores_gemma":[0.99851316,0.0008074122,0.00024624643,0.00015816174,0.00019156233,0.00008348444],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00088112993,0.0004756613,0.00056776573,0.00030042554,0.00038908888,0.00036064658,0.0009108281,0.0006249755,0.0008436044],"category_scores_gemma":[0.0031998646,0.00032121688,0.00021055859,0.00022190629,0.0008427853,0.0006843817,0.0009847714,0.0010342052,0.0001495401],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000412191,0.000051867948,0.00045640336,0.000018056611,0.000012615035,0.000033045642,0.00003781691,0.97014546,0.0023350352,0.0017829908,0.00024187559,0.02484362],"study_design_scores_gemma":[0.000005461987,0.000015469364,0.000056446825,0.0000012210224,0.0000011128822,0.0000037447865,0.0000040680216,0.9985114,0.00045924349,0.0008805836,0.000059721595,0.0000015050749],"about_ca_topic_score_codex":0.0044498793,"about_ca_topic_score_gemma":0.0028384782,"teacher_disagreement_score":0.0044498793,"about_ca_system_score_codex":0.00081340806,"about_ca_system_score_gemma":0.0010617964,"threshold_uncertainty_score":0.008847952},"labels":[],"label_agreement":null},{"id":"W4386120402","doi":"10.1137/22m1515112","title":"Satisficing Paths and Independent Multiagent Reinforcement Learning in Stochastic Games","year":2023,"lang":"en","type":"article","venue":"SIAM Journal on Mathematics of Data Science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Satisficing; Reinforcement learning; Computer science; Mathematical economics; Convergence (economics); Multi-agent system; Mathematical optimization; Mathematics; Artificial intelligence; Economics","score_opus":0.05739997469785349,"score_gpt":0.321028057988703,"score_spread":0.2636280832908495,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386120402","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03480444,0.00011177796,0.96207875,0.00030878725,0.000026178866,0.00008027321,0.00004016465,0.00012501073,0.002424588],"genre_scores_gemma":[0.84418064,0.0002507783,0.15118866,0.00020413015,0.00004158696,0.00030708834,0.00013219689,0.00007488917,0.0036200294],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981351,0.00093153823,0.00009024434,0.0003523503,0.00032002697,0.00017077875],"domain_scores_gemma":[0.99067724,0.0069721555,0.00085537456,0.00045171683,0.0006334509,0.00041005033],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029821116,0.0011690152,0.0010240049,0.0006157054,0.0006310657,0.00096375484,0.0015855868,0.0011442902,0.0022177177],"category_scores_gemma":[0.016436186,0.00066808827,0.00079061446,0.0004860193,0.002933462,0.002592826,0.0018402211,0.002411651,0.00030515945],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014992901,0.000091774134,0.001185486,0.000091885886,0.0000589416,0.000117325624,0.00024510382,0.7112673,0.0011120754,0.26565945,0.00062700245,0.019393677],"study_design_scores_gemma":[0.000032583244,0.00006424916,0.00008139953,0.000011384806,0.0000065138775,0.000021071115,0.000015659814,0.8607399,0.00046376642,0.13820365,0.00035025162,0.00000962753],"about_ca_topic_score_codex":0.0027908578,"about_ca_topic_score_gemma":0.0021951087,"teacher_disagreement_score":0.0029821116,"about_ca_system_score_codex":0.00145221,"about_ca_system_score_gemma":0.0020089305,"threshold_uncertainty_score":0.015771091},"labels":[],"label_agreement":null},{"id":"W4386126338","doi":"10.1007/s10994-023-06368-z","title":"Cautious policy programming: exploiting KL regularization for monotonic policy improvement in reinforcement learning","year":2023,"lang":"en","type":"article","venue":"Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Japan Society for the Promotion of Science","keywords":"Reinforcement learning; Monotonic function; Computer science; Mathematical optimization; Regularization (linguistics); Bellman equation; Entropy (arrow of time); Maximization; Upper and lower bounds; Exploit; Artificial intelligence; Mathematics","score_opus":0.017741848969039416,"score_gpt":0.2845911047478464,"score_spread":0.26684925577880697,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386126338","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012801321,0.00008983615,0.98547333,0.00021304436,0.000021189537,0.000033587567,0.0000073542897,0.00015192958,0.0012084505],"genre_scores_gemma":[0.75954944,0.000108295644,0.23765045,0.00033806957,0.000052175983,0.0001878915,0.000033428663,0.00013202633,0.0019482577],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99881494,0.0005271801,0.00004750833,0.00019987585,0.00029356236,0.00011687131],"domain_scores_gemma":[0.9956293,0.002981045,0.00043207465,0.00030783864,0.00043584863,0.00021398],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032194145,0.0008735629,0.0010929842,0.00055843923,0.00045680595,0.0009710325,0.0015022283,0.0013393984,0.0014394136],"category_scores_gemma":[0.011340262,0.0004942639,0.00040972698,0.00044707351,0.0022121335,0.0014884227,0.0018720459,0.0026642568,0.0002334457],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000109830966,0.000086100015,0.00067856145,0.000060223683,0.000036595295,0.000063819,0.00009729919,0.9074284,0.0031725094,0.04271054,0.000880099,0.044676077],"study_design_scores_gemma":[0.0000073822684,0.00002247797,0.000021896793,0.000004314803,0.0000015789269,0.000005402389,0.0000020475375,0.99366635,0.00036709875,0.00577041,0.00012803123,0.0000029106234],"about_ca_topic_score_codex":0.0018748858,"about_ca_topic_score_gemma":0.0014217091,"teacher_disagreement_score":0.0032194145,"about_ca_system_score_codex":0.0009406584,"about_ca_system_score_gemma":0.0016239354,"threshold_uncertainty_score":0.017026067},"labels":[],"label_agreement":null},{"id":"W4386185082","doi":"10.48550/arxiv.2308.12445","title":"An Intentional Forgetting-Driven Self-Healing Method For Deep Reinforcement Learning Systems","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Forgetting; Reinforcement learning; Convergence (economics); Adaptation (eye); Computer science; Reinforcement; Artificial intelligence; Cognitive psychology; Psychology; Social psychology; Neuroscience","score_opus":0.08574056855884084,"score_gpt":0.2481800766699188,"score_spread":0.16243950811107793,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386185082","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026865242,0.00034846726,0.96915126,0.00020853084,0.00006827913,0.000064941494,0.000023170136,0.0012549601,0.0020150214],"genre_scores_gemma":[0.88533366,0.0001374435,0.11138431,0.0002388484,0.000048392776,0.00014744622,0.000059324422,0.00011398988,0.002536396],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995493,0.00011119746,0.000035305988,0.00009861363,0.00012883174,0.00007680784],"domain_scores_gemma":[0.99912053,0.00034739982,0.00013750253,0.00010952907,0.00020415476,0.00008097916],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013027032,0.0007607979,0.00074968755,0.00038437545,0.00038198425,0.0005655109,0.0014742237,0.0007376531,0.0019666278],"category_scores_gemma":[0.002298406,0.00034787238,0.0005259843,0.00020856508,0.0006834363,0.00070309546,0.0012764747,0.0012652556,0.0002858982],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000110424306,0.00009264528,0.0013148106,0.00012033623,0.00006593963,0.00010702794,0.00014534006,0.83422965,0.007400961,0.008043576,0.0015633281,0.14680596],"study_design_scores_gemma":[0.000008112584,0.00002797519,0.000043907017,0.000003477379,0.0000046870787,0.000010038389,0.0000035539,0.9981358,0.00048236613,0.0010034188,0.0002737056,0.0000029326568],"about_ca_topic_score_codex":0.0033376224,"about_ca_topic_score_gemma":0.003331404,"teacher_disagreement_score":0.0033376224,"about_ca_system_score_codex":0.00071833364,"about_ca_system_score_gemma":0.0009254668,"threshold_uncertainty_score":0.0068894625},"labels":[],"label_agreement":null},{"id":"W4386431062","doi":"10.1007/978-3-031-43111-1_23","title":"Model-Based Policy Optimization with Neural Differential Equations for Robotic Arm Control","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Reinforcement learning; Robotic arm; Artificial intelligence; Trajectory; Process (computing); Artificial neural network; Task (project management); Node (physics); Robot; Sample (material); System dynamics; Machine learning; Engineering","score_opus":0.028591916689053084,"score_gpt":0.26111853006638336,"score_spread":0.23252661337733027,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386431062","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0038601193,0.0012646322,0.98583037,0.00031414913,0.00013318466,0.000026135454,0.000048255744,0.00021447847,0.008308584],"genre_scores_gemma":[0.78454834,0.0017451405,0.17648402,0.0003012501,0.00023930232,0.0004347606,0.0002193839,0.00032918068,0.035698567],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99976426,0.0000844682,0.00001139288,0.000040530333,0.00007459114,0.000024656427],"domain_scores_gemma":[0.9995926,0.0002625275,0.00004368441,0.000019875088,0.000065183194,0.00001604972],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00062122877,0.00088473485,0.0015322281,0.00037180842,0.0004066411,0.001055289,0.0011762426,0.0017251093,0.004027697],"category_scores_gemma":[0.001996515,0.00085532386,0.0007877132,0.00067481986,0.00091268413,0.00083953154,0.0014176767,0.001850595,0.00073219836],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000013639681,0.000016230813,0.00004254256,0.000041537965,0.00001788184,0.000012952174,0.000013264511,0.97539663,0.00032955312,0.013620737,0.0008026741,0.009692375],"study_design_scores_gemma":[0.0000024052697,0.0000039377314,0.00001297821,0.0000030294282,0.0000016452645,0.0000020996786,0.0000011080801,0.9955568,0.000044311986,0.00407883,0.00029099963,0.0000019205306],"about_ca_topic_score_codex":0.009506062,"about_ca_topic_score_gemma":0.007377015,"teacher_disagreement_score":0.009506062,"about_ca_system_score_codex":0.0012375795,"about_ca_system_score_gemma":0.0008933115,"threshold_uncertainty_score":0.018901408},"labels":[],"label_agreement":null},{"id":"W4386473849","doi":"10.1007/978-3-031-43264-4_6","title":"Exploiting Reward Machines with Deep Reinforcement Learning in Continuous Action Domains","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"York University","keywords":"Counterfactual thinking; Reinforcement learning; Computer science; Task (project management); Artificial intelligence; Action (physics); Machine learning","score_opus":0.022827126539405374,"score_gpt":0.2547727152913582,"score_spread":0.2319455887519528,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386473849","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015629109,0.0005232247,0.9802185,0.00025044003,0.00006719021,0.000017801698,0.00004003755,0.0006124904,0.002641175],"genre_scores_gemma":[0.80303293,0.000491132,0.19046871,0.00014018622,0.00008473906,0.000101892474,0.00012044504,0.0001528995,0.005407115],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99969506,0.00012215946,0.000014178769,0.00006143571,0.0000621781,0.000045013992],"domain_scores_gemma":[0.99867463,0.0009770527,0.00009475651,0.00010537728,0.000088904795,0.000059270482],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00091552566,0.0007405702,0.0010259692,0.00032160085,0.0002553188,0.0009049239,0.0013088275,0.0010851589,0.0031593917],"category_scores_gemma":[0.002806462,0.00054212764,0.00042984827,0.00043796375,0.0009273403,0.0013567287,0.0013960141,0.0019710935,0.0004883003],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000783004,0.000059136437,0.00035267157,0.00007553603,0.000037090376,0.000046952093,0.00002881074,0.8754249,0.0015460376,0.034950558,0.0016670602,0.08573295],"study_design_scores_gemma":[0.0000036875795,0.000009214962,0.000021436304,0.0000032478379,0.0000017320215,0.000004477718,0.0000012025978,0.9874893,0.0001689766,0.012143197,0.0001515091,0.0000019991264],"about_ca_topic_score_codex":0.0023960173,"about_ca_topic_score_gemma":0.002726761,"teacher_disagreement_score":0.0031593917,"about_ca_system_score_codex":0.00073288666,"about_ca_system_score_gemma":0.00063470233,"threshold_uncertainty_score":0.010569274},"labels":[],"label_agreement":null},{"id":"W4386735338","doi":"10.1007/978-3-031-43520-1_13","title":"A Relaxed Variant of Distributed Q-Learning Algorithm for Cooperative Matrix Games","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in networks and systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Rimouski","funders":"","keywords":"Algorithm; Computer science; Convergence (economics); Population-based incremental learning; Matrix (chemical analysis); Distributed algorithm; Q-learning; Weighted Majority Algorithm; Function (biology); Distributed learning; Artificial intelligence; Wake-sleep algorithm; Machine learning; Unsupervised learning; Reinforcement learning; Distributed computing; Genetic algorithm","score_opus":0.016378660899373284,"score_gpt":0.245987240694854,"score_spread":0.22960857979548072,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386735338","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0027785604,0.000066585984,0.99448156,0.00008190873,0.00006640696,0.00005724294,0.000026545045,0.00011761821,0.0023234654],"genre_scores_gemma":[0.3357725,0.00021553361,0.6492017,0.00039362794,0.00022961033,0.0006838105,0.00032263054,0.0002509998,0.0129295895],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99816906,0.00073663035,0.00007841743,0.000347598,0.00046667,0.00020162527],"domain_scores_gemma":[0.99707866,0.0016060156,0.00012262074,0.00039756604,0.0006174783,0.0001776112],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030036317,0.0011261887,0.0018629417,0.00055509806,0.0006240889,0.0012098068,0.004513896,0.0018865752,0.008530164],"category_scores_gemma":[0.006726442,0.00057085877,0.00088731566,0.0009329537,0.0016079128,0.0016952293,0.0031479748,0.0027036862,0.0017405686],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029906747,0.00017176327,0.00030279197,0.000159143,0.00007865215,0.00008501633,0.000117613694,0.78877366,0.0025383087,0.08874973,0.006064347,0.11265995],"study_design_scores_gemma":[0.000032525048,0.000037272057,0.000031535485,0.000005571968,0.0000048410884,0.00001312112,0.000005144257,0.98696125,0.00018005524,0.01212495,0.00059834024,0.0000055433734],"about_ca_topic_score_codex":0.0037780523,"about_ca_topic_score_gemma":0.002875574,"teacher_disagreement_score":0.008530164,"about_ca_system_score_codex":0.0010048785,"about_ca_system_score_gemma":0.0023050013,"threshold_uncertainty_score":0.02853626},"labels":[],"label_agreement":null},{"id":"W4386815360","doi":"10.14428/esann/2023.es2023-181","title":"Multi-Fidelity Reinforcement Learning with Control Variates","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Microsoft (Canada)","funders":"","keywords":"Reinforcement learning; Estimator; Fidelity; Control variates; Computer science; Variance (accounting); Variance reduction; High fidelity; Monte Carlo method; Mathematical optimization; Bellman equation; Artificial intelligence; Sample (material); Function (biology); Machine learning; Mathematics; Statistics; Engineering","score_opus":0.021197303717907963,"score_gpt":0.2544386487722293,"score_spread":0.23324134505432137,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386815360","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041547306,0.00020945768,0.95673484,0.0002775308,0.000032479875,0.000048575897,0.000017063017,0.00024512992,0.0008875406],"genre_scores_gemma":[0.926284,0.00006366334,0.07237108,0.00014293943,0.000030629955,0.00007874062,0.00003224056,0.000033360688,0.0009633182],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984169,0.00068899157,0.00007330183,0.00027187393,0.00041144708,0.00013744159],"domain_scores_gemma":[0.993683,0.004309967,0.0008063125,0.00048329035,0.00046541472,0.00025208178],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003771323,0.00094843376,0.001097824,0.0004180719,0.0004057766,0.0010339322,0.0014639469,0.0013072635,0.0013362812],"category_scores_gemma":[0.013707898,0.00061292277,0.00048614442,0.00031342218,0.0015560908,0.0019448781,0.002105629,0.0020378463,0.00018736227],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009155978,0.00005950141,0.0010148202,0.000031647887,0.00003058874,0.000046059176,0.000044504526,0.9782416,0.00124096,0.006766362,0.00012643542,0.012305931],"study_design_scores_gemma":[0.000008373139,0.000033754117,0.000073086834,0.0000029073528,0.0000023054656,0.0000071534073,0.0000019963827,0.99795425,0.0003402343,0.0014929926,0.0000794651,0.0000034515035],"about_ca_topic_score_codex":0.0030212898,"about_ca_topic_score_gemma":0.002345674,"teacher_disagreement_score":0.003771323,"about_ca_system_score_codex":0.0010578108,"about_ca_system_score_gemma":0.00082606595,"threshold_uncertainty_score":0.019944906},"labels":[],"label_agreement":null},{"id":"W4387394630","doi":"10.1609/aiide.v19i1.27526","title":"Herd’s Eye View: Improving Game AI Agent Learning with Collaborative Perception","year":2023,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Reinforcement learning; Perception; Computer science; Perspective (graphical); Artificial intelligence; Context (archaeology); Human–computer interaction; Psychology","score_opus":0.025858606414871924,"score_gpt":0.280315566260375,"score_spread":0.2544569598455031,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387394630","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.048231374,0.00022470178,0.9462362,0.00028817359,0.000039869476,0.00003914519,0.00002655464,0.0005999429,0.004314204],"genre_scores_gemma":[0.8604011,0.00014543664,0.13711867,0.0002073941,0.000024074901,0.00006650648,0.000053421216,0.00011747732,0.0018657737],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995396,0.00013491524,0.00001377648,0.00011955587,0.00012085169,0.000071349976],"domain_scores_gemma":[0.99903536,0.00043514196,0.000104418184,0.00018391355,0.000119322576,0.00012188633],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010212108,0.00077823986,0.0009410599,0.00037826307,0.00035328767,0.0012414219,0.0019050738,0.0011080592,0.0019529632],"category_scores_gemma":[0.0029690883,0.00049529725,0.0008316015,0.00025430642,0.00093925686,0.0022078825,0.0033643097,0.0014791848,0.00028210992],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002772543,0.0003260161,0.0041032955,0.00011746187,0.00023686033,0.0002747805,0.0007993872,0.7703824,0.021395208,0.04840927,0.0033314775,0.15034653],"study_design_scores_gemma":[0.00001704976,0.00007535093,0.00015733685,0.000004862134,0.000013507087,0.000024676488,0.000023841756,0.9903722,0.0010079141,0.007819484,0.00047564314,0.0000080799855],"about_ca_topic_score_codex":0.0036466278,"about_ca_topic_score_gemma":0.0043092687,"teacher_disagreement_score":0.0036466278,"about_ca_system_score_codex":0.0007727794,"about_ca_system_score_gemma":0.001053499,"threshold_uncertainty_score":0.0072508454},"labels":[],"label_agreement":null},{"id":"W4387423117","doi":"10.1007/978-981-99-6882-4_73","title":"LSTM-TD3-Based Control for Delayed Drone Combat Strategies","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in electrical engineering","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ministry of Education and Child Care","funders":"","keywords":"Reinforcement learning; Computer science; Battlefield; Battle; Latency (audio); Artificial intelligence; Simulation; Operations research; Engineering; Telecommunications","score_opus":0.010792545237382075,"score_gpt":0.2205415140318213,"score_spread":0.20974896879443922,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387423117","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025946567,0.00075949164,0.9606192,0.0001983992,0.00038692795,0.000037324564,0.00013607652,0.0013440228,0.010572097],"genre_scores_gemma":[0.9243574,0.0002810721,0.065616794,0.00018152244,0.00007868565,0.00009205173,0.00019373761,0.000103613966,0.009095154],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9999249,0.000010339197,0.0000057525117,0.000020627096,0.000019080402,0.000019337582],"domain_scores_gemma":[0.9998293,0.000068337744,0.000014353769,0.000013638877,0.0000633321,0.000011152723],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00028351689,0.0004982121,0.0004008485,0.00014838473,0.0002518633,0.0005846496,0.00068293425,0.00063454924,0.0049736868],"category_scores_gemma":[0.00056380295,0.0001734649,0.0003383268,0.00021677255,0.00028155235,0.00035989494,0.0004602381,0.0007934614,0.0007674229],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030127313,0.00008242329,0.00030599037,0.00023257837,0.000059200822,0.00008642707,0.00009515359,0.642592,0.025533127,0.008163553,0.0054988544,0.31704947],"study_design_scores_gemma":[0.000006889784,0.000045127093,0.00009821393,0.000008214313,0.0000063849407,0.000014377864,0.0000039042275,0.99567455,0.0020751893,0.0012139864,0.00084857724,0.000004550873],"about_ca_topic_score_codex":0.008293873,"about_ca_topic_score_gemma":0.008310479,"teacher_disagreement_score":0.008293873,"about_ca_system_score_codex":0.0005085924,"about_ca_system_score_gemma":0.00057846267,"threshold_uncertainty_score":0.016638637},"labels":[],"label_agreement":null},{"id":"W4387725412","doi":"10.48550/arxiv.2310.10502","title":"Adaptive Robot Assistance: Expertise and Influence in Multi-User Task Planning","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Partially observable Markov decision process; Computer science; Heuristics; Task (project management); Scalability; Robot; Markov decision process; Unobservable; Artificial intelligence; Process (computing); Human–computer interaction; Domain (mathematical analysis); Machine learning; Markov process; Markov chain; Engineering; Markov model; Systems engineering","score_opus":0.1377794439627749,"score_gpt":0.23352530018644124,"score_spread":0.09574585622366635,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387725412","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09250981,0.00026438604,0.90042007,0.0005592797,0.000030069588,0.00006028656,0.000025893243,0.00026667598,0.005863513],"genre_scores_gemma":[0.95780635,0.00010010821,0.040758215,0.0000656418,0.000024053457,0.000049720384,0.000019254256,0.000026743259,0.0011499567],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991345,0.0003636859,0.000025513968,0.00019073531,0.00016322083,0.00012244382],"domain_scores_gemma":[0.9974356,0.0017176734,0.00023831743,0.00016192243,0.0001865094,0.00025999875],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010212882,0.0005345059,0.0005959217,0.0004013737,0.0005634602,0.0007330962,0.0010479869,0.0009123128,0.0016485174],"category_scores_gemma":[0.005768363,0.0003928879,0.00048669294,0.00029846368,0.0011881606,0.0015199779,0.001749259,0.0012840342,0.00018686734],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002259309,0.00017766173,0.0032904902,0.000117848045,0.00006835094,0.00029586503,0.0007064675,0.8955786,0.0058736103,0.023940584,0.0009217311,0.06880288],"study_design_scores_gemma":[0.000015565487,0.00005590186,0.0005695256,0.000007785279,0.000012608057,0.00004351483,0.000045469307,0.98344284,0.000967504,0.014156882,0.0006728772,0.000009549218],"about_ca_topic_score_codex":0.0070069004,"about_ca_topic_score_gemma":0.0064493762,"teacher_disagreement_score":0.0070069004,"about_ca_system_score_codex":0.00095078483,"about_ca_system_score_gemma":0.0014375808,"threshold_uncertainty_score":0.013932228},"labels":[],"label_agreement":null},{"id":"W4387883591","doi":"10.1109/icc45041.2023.10279552","title":"Optimized Provisioning Techniques for Geo-Distributed SDP-Enabled Next Generation Networks Security","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Computer science; Provisioning; Scalability; Distributed computing; Software-defined networking; Reinforcement learning; Cloud computing; Next-generation network; Computer security; The Internet; Computer network; Artificial intelligence","score_opus":0.04171480449560785,"score_gpt":0.2806764264405227,"score_spread":0.23896162194491485,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387883591","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016094664,0.00026428932,0.9788206,0.00032716367,0.00005456702,0.00003843299,0.000034000743,0.00024325681,0.0041231103],"genre_scores_gemma":[0.8403536,0.00030154688,0.15646309,0.00010523709,0.0000350997,0.00006323871,0.00007739565,0.000076792254,0.0025241],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996834,0.000095345415,0.000013022613,0.000058595975,0.00008093149,0.000068809124],"domain_scores_gemma":[0.9995316,0.00023084955,0.000059657432,0.000057103916,0.000080639096,0.00004002717],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007837322,0.00050399784,0.0004552293,0.00033477324,0.0005152205,0.00071351003,0.0008041188,0.00056175096,0.0024221798],"category_scores_gemma":[0.0016999649,0.0003040762,0.00036015426,0.00033382062,0.00058081094,0.0009725574,0.0012105134,0.0010181793,0.000236756],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000022031256,0.000018095925,0.00027450957,0.000024805395,0.000010458286,0.00004776493,0.000039054783,0.95788556,0.0012039877,0.017939629,0.0009775967,0.021556377],"study_design_scores_gemma":[0.0000029843884,0.000005568647,0.000030208563,0.0000022908125,0.0000013872515,0.000008629603,0.000010505778,0.9937983,0.00017813782,0.00538957,0.00057096,0.0000013849501],"about_ca_topic_score_codex":0.0045295237,"about_ca_topic_score_gemma":0.0048036803,"teacher_disagreement_score":0.0045295237,"about_ca_system_score_codex":0.0009381581,"about_ca_system_score_gemma":0.0011240789,"threshold_uncertainty_score":0.009006321},"labels":[],"label_agreement":null},{"id":"W4387964087","doi":"10.48550/arxiv.2310.16686","title":"Dynamics Generalisation in Reinforcement Learning via Adaptive Context-Aware Policies","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Division of Mathematical Sciences; Canadian Institute for Advanced Research","keywords":"Adapter (computing); Reinforcement learning; Computer science; Architecture; Artificial intelligence; Robot; Context (archaeology); Machine learning; Human–computer interaction","score_opus":0.10101501291661223,"score_gpt":0.20956679359048505,"score_spread":0.10855178067387282,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387964087","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1011084,0.00030803905,0.89490914,0.0002442329,0.000038182596,0.00008707371,0.000028258482,0.0009842643,0.0022923767],"genre_scores_gemma":[0.95633465,0.00011150909,0.042044234,0.00009144207,0.00001812703,0.00008686996,0.000033877543,0.00003989166,0.0012394112],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994814,0.00015951767,0.00003183087,0.00016361885,0.0000968022,0.00006693045],"domain_scores_gemma":[0.99890435,0.00056257367,0.00013915176,0.00016178767,0.00017183638,0.00006021535],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012806132,0.0007699578,0.00072527246,0.00034217778,0.0002532126,0.00056835805,0.0010196442,0.000756811,0.0012109719],"category_scores_gemma":[0.0050173085,0.00038766622,0.00041872382,0.00022873469,0.0010200478,0.0011069872,0.0011889815,0.0014171967,0.00019011526],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000083606676,0.000061969185,0.0012148162,0.000052541123,0.000044534725,0.00007840126,0.00015954705,0.9236906,0.0038163364,0.0068659834,0.00026850242,0.0636632],"study_design_scores_gemma":[0.000009732719,0.000032060212,0.0001463286,0.0000044947988,0.0000059844947,0.000009085841,0.0000052072196,0.995169,0.000516945,0.003931563,0.00016613731,0.0000035064465],"about_ca_topic_score_codex":0.005822123,"about_ca_topic_score_gemma":0.0042525777,"teacher_disagreement_score":0.005822123,"about_ca_system_score_codex":0.00084565306,"about_ca_system_score_gemma":0.00078007526,"threshold_uncertainty_score":0.011576474},"labels":[],"label_agreement":null},{"id":"W4388040445","doi":"10.1109/pimrc56721.2023.10293845","title":"DRJLRA: A Deep Reinforcement Learning-Based Joint Load and Resource Allocation in Heterogeneous Coded Distributed Computing","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Alberta Innovates","keywords":"Computer science; Benchmark (surveying); Reinforcement learning; Distributed computing; Computation; Task (project management); Set (abstract data type); Resource allocation; Resource management (computing); Scheme (mathematics); Artificial intelligence; Theoretical computer science; Algorithm; Computer network","score_opus":0.020413338865493815,"score_gpt":0.24458704463290457,"score_spread":0.22417370576741075,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388040445","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027503189,0.00053297647,0.9671398,0.00034380477,0.00010184913,0.00006531188,0.000041495277,0.001289142,0.002982452],"genre_scores_gemma":[0.8456373,0.00017413778,0.1493637,0.00033266598,0.000051737592,0.00013393948,0.000104390376,0.0001405817,0.004061595],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994497,0.00014835874,0.00002472471,0.00012535404,0.00013736481,0.00011445551],"domain_scores_gemma":[0.99907696,0.0004148808,0.000102706304,0.00009627032,0.00019594861,0.0001133176],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001172718,0.0007457006,0.00096071576,0.00032663284,0.00040151202,0.00082291313,0.0021434084,0.00094761443,0.0018726964],"category_scores_gemma":[0.0026856202,0.00036284197,0.00032429578,0.00032146965,0.00082606083,0.0010662336,0.0013363583,0.0014949342,0.0004323287],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010529431,0.00008686412,0.0005619286,0.00006353945,0.000032238328,0.000046614627,0.000044502558,0.9262871,0.002165774,0.0041167545,0.0017574584,0.06473191],"study_design_scores_gemma":[0.000007446173,0.0000171144,0.000025921849,0.0000027306316,0.0000020159857,0.0000045568795,0.000003173194,0.99878246,0.00026579257,0.00066509645,0.00022155931,0.0000021757273],"about_ca_topic_score_codex":0.006917445,"about_ca_topic_score_gemma":0.0075553423,"teacher_disagreement_score":0.006917445,"about_ca_system_score_codex":0.0010215596,"about_ca_system_score_gemma":0.0020369813,"threshold_uncertainty_score":0.013754368},"labels":[],"label_agreement":null},{"id":"W4388191662","doi":"10.1016/j.asoc.2023.110975","title":"Ensemble reinforcement learning: A survey","year":2023,"lang":"en","type":"article","venue":"Applied Soft Computing","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Science and Technology Innovation Group of Shanxi Province; National Natural Science Foundation of China","keywords":"Reinforcement learning; Computer science; Popularity; Generalization; Field (mathematics); Ensemble learning; Artificial intelligence; Machine learning; Open research; Selection (genetic algorithm); Data science","score_opus":0.03247530673738754,"score_gpt":0.2652782368121937,"score_spread":0.23280293007480615,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388191662","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010999573,0.27910146,0.69361275,0.0012565699,0.00060919946,0.00012284386,0.0001495036,0.00048359824,0.013664541],"genre_scores_gemma":[0.35036543,0.3429073,0.29028913,0.0009503708,0.0031284497,0.0004454179,0.00079465425,0.0003378477,0.010781428],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99881786,0.00031470685,0.00012106516,0.00030243164,0.0003819614,0.00006187928],"domain_scores_gemma":[0.99737203,0.001719424,0.000091810194,0.00023162492,0.0005101206,0.00007510272],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002415697,0.00122857,0.0031333105,0.0013902129,0.00041765766,0.0018445143,0.0020288166,0.0012901014,0.002254658],"category_scores_gemma":[0.0040550986,0.0006714781,0.0009979479,0.0035513295,0.00076051726,0.002308961,0.0016232607,0.0016868559,0.0008028322],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000065956905,0.00025625664,0.0017572547,0.0012258962,0.0002151256,0.000030611503,0.00007066156,0.09399552,0.0006608848,0.025306001,0.0055519575,0.8708639],"study_design_scores_gemma":[0.000042855376,0.0003999245,0.0021625517,0.00059614744,0.00020005989,0.00029355288,0.00011872242,0.853248,0.0021623177,0.080418475,0.060277257,0.000080044825],"about_ca_topic_score_codex":0.0025784774,"about_ca_topic_score_gemma":0.002215151,"teacher_disagreement_score":0.0031333105,"about_ca_system_score_codex":0.0007073669,"about_ca_system_score_gemma":0.0011392821,"threshold_uncertainty_score":0.0127756},"labels":[],"label_agreement":null},{"id":"W4388342252","doi":"10.1002/aisy.202300352","title":"A Multiobjective Collaborative Deep Reinforcement Learning Algorithm for Jumping Optimization of Bipedal Robot","year":2023,"lang":"en","type":"article","venue":"Advanced Intelligent Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"Government of Jiangsu Province; Natural Science Foundation of Jiangsu Province; National Natural Science Foundation of China","keywords":"Reinforcement learning; Computer science; Markov decision process; Convergence (economics); Robot; Jumping; Artificial intelligence; Q-learning; Nonlinear system; Robotics; Mathematical optimization; Machine learning; Markov process; Mathematics","score_opus":0.020510999090521592,"score_gpt":0.28889894933451854,"score_spread":0.268387950243997,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388342252","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02068157,0.0002190216,0.9761524,0.00013633708,0.00004380432,0.000040416595,0.000022616176,0.00046843736,0.0022353064],"genre_scores_gemma":[0.8224848,0.0001322884,0.17239082,0.00017530215,0.000030164285,0.00018994058,0.000098922996,0.000078612524,0.004419071],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998305,0.00003396046,0.000009656935,0.00004380692,0.000044731212,0.000037366346],"domain_scores_gemma":[0.9996773,0.00015606316,0.000042876258,0.000018827835,0.00007087359,0.000033999262],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00063156965,0.0007091839,0.0008720791,0.00030922712,0.00032626,0.00048337292,0.0009357219,0.00093008694,0.002127284],"category_scores_gemma":[0.001070868,0.00036263865,0.00043236616,0.00023720291,0.000462358,0.00044139702,0.0009501291,0.0011018914,0.00025572247],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004609089,0.00003609782,0.00038463235,0.000040836327,0.000029915329,0.000055219367,0.000034172066,0.94641316,0.0018654871,0.003052241,0.0007858797,0.047256354],"study_design_scores_gemma":[0.0000047595563,0.000011912214,0.000021792841,0.0000019659137,0.000001961883,0.0000035449273,0.0000014547534,0.9993457,0.000130428,0.00038227733,0.00009279848,0.000001416131],"about_ca_topic_score_codex":0.0065235747,"about_ca_topic_score_gemma":0.0047508106,"teacher_disagreement_score":0.0065235747,"about_ca_system_score_codex":0.0005820642,"about_ca_system_score_gemma":0.0010697239,"threshold_uncertainty_score":0.012971222},"labels":[],"label_agreement":null},{"id":"W4388483498","doi":"10.1109/ase56229.2023.00121","title":"An Intentional Forgetting-Driven Self-Healing Method for Deep Reinforcement Learning Systems","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Forgetting; Reinforcement learning; Convergence (economics); Computer science; Adaptation (eye); Reinforcement; Artificial intelligence; Cognitive psychology; Engineering; Psychology","score_opus":0.02687934710323557,"score_gpt":0.31674026995211785,"score_spread":0.28986092284888226,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388483498","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02624189,0.00034337162,0.9699469,0.0002062104,0.000065567445,0.00006636362,0.000023422956,0.0012727738,0.0018334185],"genre_scores_gemma":[0.8782208,0.00014014693,0.118468806,0.0002474907,0.000048783797,0.00015304628,0.00006331343,0.00011720595,0.0025403965],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995134,0.00012197407,0.000038980645,0.00010828959,0.00013739649,0.00008001544],"domain_scores_gemma":[0.99905616,0.00037140644,0.00014565764,0.000120505014,0.00021994655,0.000086261614],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013800845,0.0007548188,0.0007451241,0.00040227233,0.00037985845,0.00057359884,0.0015502127,0.0007581536,0.0019423364],"category_scores_gemma":[0.002420658,0.00034926742,0.0005263742,0.00021601697,0.00069391157,0.0007412296,0.0013063673,0.001278333,0.00028236824],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011208328,0.00009461701,0.001361753,0.00012270325,0.00006613274,0.00010488691,0.0001473168,0.8266681,0.0070727332,0.00784336,0.001588657,0.15481766],"study_design_scores_gemma":[0.000008782428,0.000029539955,0.00004717047,0.0000035936546,0.0000048848583,0.000010583049,0.0000036929503,0.9980411,0.0005048758,0.0010537462,0.00028899935,0.0000030849694],"about_ca_topic_score_codex":0.0032715704,"about_ca_topic_score_gemma":0.0032872644,"teacher_disagreement_score":0.0032715704,"about_ca_system_score_codex":0.0007441041,"about_ca_system_score_gemma":0.00094398204,"threshold_uncertainty_score":0.007298708},"labels":[],"label_agreement":null},{"id":"W4388551180","doi":"10.1007/s10458-023-09628-3","title":"ASN: action semantics network for multiagent reinforcement learning","year":2023,"lang":"en","type":"article","venue":"Autonomous Agents and Multi-Agent Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Semantics (computer science); Artificial intelligence; Action (physics); Action selection; Artificial neural network; Multi-agent system; Programming language; Perception","score_opus":0.08029373040729547,"score_gpt":0.3143184978309977,"score_spread":0.2340247674237022,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388551180","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003725223,0.00012398913,0.98666763,0.000193353,0.00014200898,0.00008172087,0.0007976446,0.005330429,0.0029378557],"genre_scores_gemma":[0.36841822,0.00041364305,0.61817014,0.00034918095,0.00008897608,0.0006619643,0.0026344056,0.00089380477,0.008369761],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99963856,0.00012609792,0.00002832627,0.00007360692,0.00009893697,0.000034459776],"domain_scores_gemma":[0.99928147,0.0003043014,0.000054362215,0.00012709494,0.00016594527,0.00006689933],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009295625,0.00060987694,0.00064889865,0.0006783214,0.00042676504,0.0009724643,0.0012268673,0.0008524017,0.0069792713],"category_scores_gemma":[0.0030703505,0.00030919057,0.0005946842,0.00052339886,0.00059078645,0.0014271255,0.0010334615,0.0014778248,0.0013917078],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00064214406,0.00021848206,0.0010427422,0.00033065054,0.000088133835,0.0001942069,0.00012415854,0.5040024,0.004956281,0.19107346,0.02917764,0.2681497],"study_design_scores_gemma":[0.00003111059,0.000028762386,0.000095887975,0.000015900527,0.000013736816,0.00003266247,0.000010255721,0.89956784,0.0016537757,0.08935808,0.00918216,0.000009831511],"about_ca_topic_score_codex":0.0058868504,"about_ca_topic_score_gemma":0.00775491,"teacher_disagreement_score":0.0069792713,"about_ca_system_score_codex":0.0010554262,"about_ca_system_score_gemma":0.0015514717,"threshold_uncertainty_score":0.023347914},"labels":[],"label_agreement":null},{"id":"W4388741154","doi":"10.1016/j.engappai.2023.107518","title":"Reinforcement learning to achieve real-time control of triple inverted pendulum","year":2023,"lang":"en","type":"article","venue":"Engineering Applications of Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Reinforcement learning; Inverted pendulum; Markov decision process; Process (computing); Convergence (economics); Sample (material); Artificial intelligence; Machine learning; Markov process","score_opus":0.01839544774908922,"score_gpt":0.26089423124250033,"score_spread":0.2424987834934111,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388741154","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15770893,0.00028304977,0.8333761,0.00028806637,0.00015583802,0.00007482098,0.00001747884,0.00046359748,0.007632162],"genre_scores_gemma":[0.9872322,0.000031602616,0.011602793,0.000030269643,0.0000088976885,0.000039333652,0.000008235485,0.000010344459,0.0010362825],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99980944,0.000050683528,0.000010135874,0.000029481169,0.00005297807,0.000047406742],"domain_scores_gemma":[0.9995547,0.00019283533,0.000055915178,0.00002464035,0.00012219125,0.000049843755],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008345326,0.000525987,0.00063207815,0.00025751148,0.0004483421,0.0004742176,0.0006378268,0.0006402639,0.0016340254],"category_scores_gemma":[0.0014852457,0.0002655919,0.00028648454,0.00017829766,0.0006889307,0.00028501343,0.00073102536,0.00067282043,0.00016875956],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012011928,0.00007299136,0.00038843692,0.000060108956,0.000027765445,0.000120269455,0.00008235272,0.9503579,0.00627796,0.009027278,0.00064765516,0.032817133],"study_design_scores_gemma":[0.000010161038,0.000049034057,0.000056870686,0.0000023639466,0.0000025058516,0.000008133452,0.0000029374319,0.9985965,0.0003793714,0.0007863311,0.00010322887,0.0000025141435],"about_ca_topic_score_codex":0.0067387666,"about_ca_topic_score_gemma":0.003638842,"teacher_disagreement_score":0.0067387666,"about_ca_system_score_codex":0.0005346378,"about_ca_system_score_gemma":0.00071711856,"threshold_uncertainty_score":0.013399065},"labels":[],"label_agreement":null},{"id":"W4388761312","doi":"10.17760/d20486919","title":"Bayesian partially observable reinforcement learning","year":2023,"lang":"en","type":"dissertation","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Science North","funders":"","keywords":"Computer science; Reinforcement learning; Inference; State (computer science); Bayesian inference; Artificial intelligence; Bayesian probability","score_opus":0.025291122515353624,"score_gpt":0.27267902967139845,"score_spread":0.24738790715604483,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388761312","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026886713,0.0011119677,0.9551683,0.0013516194,0.00017860194,0.00014818388,0.00051951525,0.0011394434,0.0134956045],"genre_scores_gemma":[0.88336045,0.00062309037,0.10398153,0.0005080828,0.0001272012,0.00035850378,0.00073716807,0.00010172516,0.010202217],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.998489,0.0006818589,0.000066106055,0.0002989877,0.00026958119,0.00019438994],"domain_scores_gemma":[0.9939931,0.004361428,0.00043764102,0.00030345746,0.00058779103,0.0003165438],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021155153,0.0014098691,0.0022658673,0.00072344387,0.00059589016,0.0014628745,0.00214481,0.0021127479,0.005781085],"category_scores_gemma":[0.0106974505,0.0007594335,0.00064040575,0.00073464186,0.0016604237,0.001903733,0.0015293356,0.0023429617,0.0010140714],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020691416,0.000104818304,0.0016280485,0.0001584724,0.00007823301,0.00014168274,0.00010664282,0.8927065,0.0003054691,0.059371796,0.004042673,0.04114872],"study_design_scores_gemma":[0.000034910136,0.000020897487,0.0001382384,0.000013140371,0.000008769108,0.00001227131,0.0000087758335,0.9643597,0.000080999685,0.034483973,0.0008287118,0.00000960482],"about_ca_topic_score_codex":0.0106453225,"about_ca_topic_score_gemma":0.0112309605,"teacher_disagreement_score":0.0106453225,"about_ca_system_score_codex":0.0020484067,"about_ca_system_score_gemma":0.002154075,"threshold_uncertainty_score":0.021166742},"labels":[],"label_agreement":null},{"id":"W4388820294","doi":"10.1109/tie.2023.3331074","title":"Model-Based Reinforcement Learning With Probabilistic Ensemble Terminal Critics for Data-Efficient Control Applications","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Industrial Electronics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"National Research Foundation of Korea","keywords":"Reinforcement learning; Probabilistic logic; Computer science; Trajectory; Artificial intelligence; Machine learning","score_opus":0.06410995923129628,"score_gpt":0.2901327633149109,"score_spread":0.2260228040836146,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388820294","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0121052805,0.00021288953,0.98592144,0.00010735818,0.000027817036,0.000022032784,0.000021910395,0.00045629946,0.0011249888],"genre_scores_gemma":[0.8995781,0.00015958732,0.09834104,0.00010565727,0.000034063793,0.00011008895,0.000096130876,0.000077695775,0.0014975796],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99951375,0.0001254161,0.000030095733,0.00009318445,0.0001784805,0.000059008653],"domain_scores_gemma":[0.9985915,0.00075654715,0.00017833964,0.000121797784,0.00029040975,0.000061384395],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013751397,0.00089230726,0.0010370298,0.00038655946,0.00030927162,0.00078834244,0.0013022387,0.00081755884,0.0013917219],"category_scores_gemma":[0.0042218477,0.0004540295,0.00048295697,0.00034992697,0.00081217085,0.00080238405,0.001073225,0.0019201328,0.0003111416],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000041203697,0.00002055082,0.0003499947,0.000028185787,0.000019560848,0.000027689564,0.000024189001,0.9746302,0.0009101942,0.0030741785,0.00033475808,0.02053928],"study_design_scores_gemma":[0.000002717919,0.000009476983,0.000021724023,0.0000019932945,0.000001957272,0.0000036954918,8.6477115e-7,0.99899083,0.0001898999,0.00068714155,0.00008799682,0.0000016904485],"about_ca_topic_score_codex":0.0046041906,"about_ca_topic_score_gemma":0.003674057,"teacher_disagreement_score":0.0046041906,"about_ca_system_score_codex":0.0007451483,"about_ca_system_score_gemma":0.0011975841,"threshold_uncertainty_score":0.009154797},"labels":[],"label_agreement":null},{"id":"W4388904319","doi":"10.1016/j.ifacol.2023.10.924","title":"Reinforcement Learning with Partial Parametric Model Knowledge","year":2023,"lang":"en","type":"article","venue":"IFAC-PapersOnLine","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Honeywell (Canada); University of British Columbia","funders":"","keywords":"Reinforcement learning; Computer science; Ignorance; Bridge (graph theory); Parametric statistics; Control (management); Linear-quadratic regulator; Complete information; Partial least squares regression; Mathematical optimization; Artificial intelligence; Mathematics; Machine learning; Mathematical economics; Statistics; Law","score_opus":0.031056413427376525,"score_gpt":0.27725972538507326,"score_spread":0.24620331195769674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388904319","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0069299666,0.00012014721,0.9911972,0.00013992457,0.00002474205,0.000019998482,0.000015664538,0.00025925724,0.0012930543],"genre_scores_gemma":[0.8733368,0.00021134678,0.12312568,0.00016956928,0.00007096744,0.00015187109,0.00007083255,0.000068067566,0.0027948676],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99895847,0.00036764675,0.000043883818,0.00020183284,0.00032766955,0.00010034593],"domain_scores_gemma":[0.9969639,0.0018619554,0.00033405618,0.00041976522,0.00028918093,0.00013116117],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017450907,0.0010534142,0.0012260268,0.00040942404,0.00030373028,0.0010753127,0.001804298,0.000984277,0.0018801374],"category_scores_gemma":[0.006135347,0.00050296483,0.00073504064,0.00044108808,0.0016231579,0.001856481,0.0016748114,0.001888555,0.0003622758],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000063257656,0.000056317167,0.00032121912,0.000076827484,0.000052530027,0.000081117294,0.00006784549,0.93941355,0.0012625929,0.020082043,0.0004964265,0.038026232],"study_design_scores_gemma":[0.000011209476,0.00003946074,0.000035088946,0.0000055594537,0.000005421701,0.000010582114,0.0000029947648,0.9901736,0.00037095102,0.009029055,0.00030977625,0.0000062457993],"about_ca_topic_score_codex":0.002697314,"about_ca_topic_score_gemma":0.0021732065,"teacher_disagreement_score":0.002697314,"about_ca_system_score_codex":0.00059631857,"about_ca_system_score_gemma":0.0012650755,"threshold_uncertainty_score":0.009229004},"labels":[],"label_agreement":null},{"id":"W4389218123","doi":"10.48550/arxiv.2311.17855","title":"Maximum Entropy Model Correction in Reinforcement Learning","year":2023,"lang":"en","type":"preprint","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Government of Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Principle of maximum entropy; Computer science; Convergence (economics); Bellman equation; Entropy (arrow of time); Algorithm; Mathematical optimization; Function (biology); Applied mathematics; Mathematics; Artificial intelligence","score_opus":0.021591920548086455,"score_gpt":0.25046835554405456,"score_spread":0.2288764349959681,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389218123","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0064599975,0.00009126473,0.9921814,0.0001326908,0.000019451947,0.000013849568,0.000009632314,0.00012409576,0.0009676352],"genre_scores_gemma":[0.80875224,0.00020291637,0.18759719,0.0001433172,0.000054219738,0.00015585923,0.000050721592,0.00011511029,0.002928525],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99874,0.000575661,0.00003229792,0.00017237234,0.00037645848,0.00010319754],"domain_scores_gemma":[0.99585766,0.0029907795,0.000347722,0.00033754227,0.00034744886,0.00011882331],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002009905,0.00078265864,0.0011096838,0.0005010835,0.0005208349,0.0010168343,0.0015695423,0.0011162168,0.0015824023],"category_scores_gemma":[0.010339837,0.0005214434,0.00057473197,0.00047647973,0.0020378414,0.0019507005,0.0014410287,0.0017565407,0.00026034052],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003930961,0.000021024449,0.00034753472,0.000039452738,0.000020419886,0.000047201185,0.000051855342,0.9274867,0.0005727522,0.056977708,0.00033021247,0.014065852],"study_design_scores_gemma":[0.000003525278,0.000012601471,0.000024923374,0.0000035994665,0.0000019295649,0.0000069616826,0.0000020590278,0.98434097,0.00022530086,0.0152191995,0.00015579026,0.0000030916947],"about_ca_topic_score_codex":0.0039727734,"about_ca_topic_score_gemma":0.0028558075,"teacher_disagreement_score":0.0039727734,"about_ca_system_score_codex":0.001525466,"about_ca_system_score_gemma":0.00165307,"threshold_uncertainty_score":0.011068046},"labels":[],"label_agreement":null},{"id":"W4389339279","doi":"10.1145/3618338","title":"Scene-Aware Activity Program Generation with Language Guidance","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Graphics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"National Natural Science Foundation of China","keywords":"Computer science; Executable; Generalizability theory; Encoder; Scene graph; Graph; Artificial intelligence; Rationality; Perception; Representation (politics); Key (lock); Machine learning; Theoretical computer science; Human–computer interaction; Natural language processing; Programming language","score_opus":0.03607292347747975,"score_gpt":0.29635666487959256,"score_spread":0.2602837414021128,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389339279","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04224094,0.00036692974,0.92591864,0.00037562373,0.00007016474,0.00020856046,0.0009203668,0.026294598,0.0036041895],"genre_scores_gemma":[0.35633945,0.00018761518,0.63278157,0.0002892415,0.000036958845,0.00031955366,0.0036993246,0.0013120748,0.005034208],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996137,0.00007108535,0.00001757688,0.00015792315,0.00009441622,0.000045309476],"domain_scores_gemma":[0.9993742,0.00025160777,0.000046971276,0.00017582584,0.00011081447,0.000040663395],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000386992,0.0009515265,0.0005131109,0.0005793764,0.00025712873,0.00060768425,0.001753979,0.0006410418,0.0022349556],"category_scores_gemma":[0.0018440965,0.00028843808,0.00074596034,0.00042353314,0.0005531499,0.0011053291,0.00086495554,0.0012328394,0.0010862699],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00050159515,0.0005344068,0.005020841,0.00043897124,0.000078434925,0.0003113825,0.00035191677,0.21084848,0.03753395,0.0134367645,0.0224297,0.7085135],"study_design_scores_gemma":[0.000049452738,0.00007513407,0.00031579036,0.000011035829,0.00001644271,0.00006501004,0.00003082643,0.97349316,0.012476387,0.008492665,0.0049622175,0.000011938836],"about_ca_topic_score_codex":0.0053333617,"about_ca_topic_score_gemma":0.011451673,"teacher_disagreement_score":0.0053333617,"about_ca_system_score_codex":0.00060036866,"about_ca_system_score_gemma":0.0014192072,"threshold_uncertainty_score":0.01060462},"labels":[],"label_agreement":null},{"id":"W4389395619","doi":"10.1007/s10458-023-09607-8","title":"Uniformly constrained reinforcement learning","year":2023,"lang":"en","type":"article","venue":"Autonomous Agents and Multi-Agent Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Exploratory Research for Advanced Technology","keywords":"Reinforcement learning; Markov decision process; Mathematical optimization; Operator (biology); Counterexample; Probabilistic logic; Context (archaeology); Computer science; Heuristic; Constraint (computer-aided design); Monotonic function; Markov process; Mathematics; Artificial intelligence","score_opus":0.04139100013431038,"score_gpt":0.2750660092176482,"score_spread":0.2336750090833378,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389395619","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027625462,0.0006804053,0.95548433,0.00050969626,0.00012214326,0.00006332522,0.00011390369,0.00041219307,0.014988574],"genre_scores_gemma":[0.9422298,0.00031267013,0.046350304,0.00021169982,0.00006872074,0.00015169264,0.00017928921,0.00009292757,0.010402894],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99935514,0.00022258733,0.000029895253,0.00017270369,0.00013540163,0.000084188236],"domain_scores_gemma":[0.9980286,0.0012385149,0.00014817098,0.0001806703,0.00025844286,0.00014573056],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00096506457,0.0008828427,0.0011022814,0.00037454168,0.00039079922,0.0010353462,0.0010017859,0.0011899734,0.006524146],"category_scores_gemma":[0.0058326735,0.00040598327,0.0003059603,0.0004668394,0.0013304302,0.0014364108,0.0018903662,0.0012665386,0.00061807776],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023686828,0.00008823974,0.00043204834,0.00013291732,0.000044533703,0.000085458036,0.000042510004,0.84270465,0.0015195337,0.09946688,0.00337268,0.05187372],"study_design_scores_gemma":[0.000020271702,0.00003177453,0.00007608325,0.00000794145,0.000004967699,0.000013111182,0.0000042866536,0.978216,0.00033219502,0.020699125,0.00058966264,0.0000046621617],"about_ca_topic_score_codex":0.0031834473,"about_ca_topic_score_gemma":0.0022853785,"teacher_disagreement_score":0.006524146,"about_ca_system_score_codex":0.0009976408,"about_ca_system_score_gemma":0.0009828283,"threshold_uncertainty_score":0.021825433},"labels":[],"label_agreement":null},{"id":"W4389482477","doi":"10.54254/2753-8818/19/20230542","title":"An evaluation of reinforcement learning performance in the iterated prisoner’s dilemma","year":2023,"lang":"en","type":"article","venue":"Theoretical and Natural Science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"University of Cambridge","keywords":"Reinforcement learning; Dilemma; Iterated function; Prisoner's dilemma; Reinforcement; Computer science; Artificial neural network; Artificial intelligence; Psychology; Social psychology; Mathematics","score_opus":0.01822272713279608,"score_gpt":0.29486063010059416,"score_spread":0.27663790296779805,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389482477","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96886873,0.00042408574,0.025430014,0.00017104995,0.00005887805,0.0001041535,0.000049106424,0.00028869155,0.00460524],"genre_scores_gemma":[0.9924469,0.00007353564,0.0068222876,0.000020019457,0.00000432404,0.000043726017,0.000041118725,0.000010572577,0.0005373163],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989033,0.0005034307,0.00009599067,0.00015482295,0.00022741075,0.00011506144],"domain_scores_gemma":[0.9943768,0.0036928295,0.0005878732,0.0002830663,0.0006777836,0.00038175355],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033451002,0.000816157,0.0007883237,0.0004783937,0.00033568952,0.0006366331,0.00078205176,0.0009077642,0.00096678245],"category_scores_gemma":[0.009292775,0.00017181729,0.00026894297,0.00025160456,0.0005251214,0.0007712796,0.00058602536,0.0007163399,0.00015685457],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018213256,0.0022725638,0.013580934,0.00044353216,0.0003370568,0.00022612105,0.00029775366,0.8819666,0.010531082,0.003682642,0.00094249373,0.08389793],"study_design_scores_gemma":[0.000074228155,0.0019457971,0.0024282602,0.000014674493,0.00002949631,0.00004025367,0.000053275347,0.9903127,0.0037435323,0.0011245918,0.00021221998,0.000020948437],"about_ca_topic_score_codex":0.0036236236,"about_ca_topic_score_gemma":0.0019430788,"teacher_disagreement_score":0.0036236236,"about_ca_system_score_codex":0.0007840127,"about_ca_system_score_gemma":0.00055920274,"threshold_uncertainty_score":0.017690778},"labels":[],"label_agreement":null},{"id":"W4389551935","doi":"10.3390/su152416741","title":"Harnessing Online Knowledge Transfer for Enhanced Search and Rescue Decisions via Multi-Agent Reinforcement Learning","year":2023,"lang":"en","type":"article","venue":"Sustainability","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Reinforcement learning; Computer science; Benchmark (surveying); Transformative learning; Process (computing); Artificial intelligence; Internet of Things; Machine learning; Computer security","score_opus":0.05112242449648516,"score_gpt":0.35154827041862424,"score_spread":0.30042584592213906,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389551935","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07950896,0.00040596016,0.91563064,0.0004790988,0.00007299471,0.00007744125,0.000038113867,0.0005704836,0.003216281],"genre_scores_gemma":[0.9589984,0.00009077794,0.03980453,0.00012824347,0.000020951271,0.00006343646,0.00003248167,0.000024571491,0.00083663577],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995945,0.00015101241,0.000020095298,0.000085998334,0.000081797196,0.00006657133],"domain_scores_gemma":[0.99819285,0.0011730359,0.00020320294,0.00011891694,0.00019193091,0.00012002692],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013111439,0.0008636704,0.0008620803,0.0003241304,0.00028936658,0.00078457623,0.001112735,0.0009230905,0.0012995525],"category_scores_gemma":[0.0047648274,0.0003450451,0.00034191122,0.00024059102,0.0009073828,0.0010552694,0.001315961,0.0013012809,0.00022768245],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000057223984,0.00009363871,0.00084553467,0.000046294175,0.000035945453,0.000065348475,0.000044605415,0.9663215,0.0014047502,0.0042002215,0.00040642192,0.026478494],"study_design_scores_gemma":[0.000006447051,0.000022156572,0.000044745128,0.0000027569683,0.0000030711008,0.000004669718,0.000003108544,0.9981719,0.0001879528,0.0014535694,0.00009728765,0.000002322623],"about_ca_topic_score_codex":0.0032593932,"about_ca_topic_score_gemma":0.0029690738,"teacher_disagreement_score":0.0032593932,"about_ca_system_score_codex":0.0006284784,"about_ca_system_score_gemma":0.0011702497,"threshold_uncertainty_score":0.0069340467},"labels":[],"label_agreement":null},{"id":"W4389650673","doi":"10.48550/arxiv.2312.05822","title":"Toward Open-ended Embodied Tasks Solving","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Microsoft Research","keywords":"Embodied cognition; Task (project management); Computer science; Robot; Human–computer interaction; Artificial intelligence; Control (management); State (computer science); Plan (archaeology); Engineering","score_opus":0.19266900952023353,"score_gpt":0.23433942826683515,"score_spread":0.04167041874660163,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389650673","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04031616,0.00019797303,0.95404774,0.00040265277,0.000020039122,0.000054976328,0.000019681689,0.0004364132,0.004504372],"genre_scores_gemma":[0.6236738,0.0002897148,0.37175715,0.00014407869,0.000016672007,0.00016442631,0.00007766831,0.00011148626,0.0037650575],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99944276,0.00023694115,0.000027790067,0.00010858203,0.00013495273,0.00004892066],"domain_scores_gemma":[0.99877995,0.0007518661,0.0001095613,0.00014581066,0.00010054524,0.00011230418],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011932481,0.00065960456,0.00046513503,0.00024310325,0.00038941877,0.0010680782,0.0011381947,0.001217651,0.001634959],"category_scores_gemma":[0.0037990708,0.0003451831,0.00054754736,0.00020078849,0.0019140862,0.0018542522,0.0031342604,0.0020751513,0.00029526747],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011960878,0.0002122792,0.0012144967,0.00026676687,0.000050498402,0.00017561596,0.0018155169,0.6963977,0.019832298,0.1536111,0.0015638932,0.12474013],"study_design_scores_gemma":[0.000032808188,0.00006546283,0.00013526302,0.000024226192,0.000008128227,0.00003485316,0.00012517397,0.8921994,0.0036456338,0.10014428,0.0035731168,0.000011615622],"about_ca_topic_score_codex":0.0014209171,"about_ca_topic_score_gemma":0.0017534897,"teacher_disagreement_score":0.001634959,"about_ca_system_score_codex":0.0006255055,"about_ca_system_score_gemma":0.0008105407,"threshold_uncertainty_score":0.0063106418},"labels":[],"label_agreement":null},{"id":"W4389665442","doi":"10.1109/iros55552.2023.10342281","title":"Asynchronous, Option-Based Multi-Agent Policy Gradient: A Conditional Reasoning Approach","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); Simon Fraser University","funders":"","keywords":"Asynchronous communication; Computer science; Action (physics); State (computer science); Artificial intelligence; Machine learning; Algorithm","score_opus":0.03117581096845054,"score_gpt":0.28274496944238203,"score_spread":0.25156915847393146,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389665442","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008815111,0.00013179798,0.9880802,0.00029309953,0.00002992088,0.00006345919,0.000054546977,0.00046969933,0.0020622239],"genre_scores_gemma":[0.70844305,0.00014764175,0.28810468,0.0003034717,0.000067819055,0.00021784712,0.00016171695,0.00014367883,0.0024101443],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99865675,0.0004952818,0.00007084642,0.00025101146,0.00038375045,0.0001424287],"domain_scores_gemma":[0.9955129,0.0029769526,0.0003874021,0.00028978602,0.0005428832,0.00028999284],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00380785,0.001061407,0.0012515647,0.00081604184,0.0006778674,0.0012133429,0.003059263,0.0016337243,0.0039948463],"category_scores_gemma":[0.008180939,0.0006841342,0.00089072686,0.0005812105,0.0015892067,0.002050728,0.0019943623,0.0023352248,0.0005352071],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014440273,0.00008779727,0.00070202886,0.00009248219,0.0000472767,0.00013027871,0.00011537774,0.9159421,0.0014291864,0.037944347,0.0014320966,0.04193249],"study_design_scores_gemma":[0.000012330604,0.000013604148,0.000038880287,0.0000046504265,0.0000045306847,0.000008393222,0.000004252472,0.99230266,0.00022555093,0.0071671633,0.00021332037,0.00000472841],"about_ca_topic_score_codex":0.005149842,"about_ca_topic_score_gemma":0.0049713934,"teacher_disagreement_score":0.005149842,"about_ca_system_score_codex":0.0014835714,"about_ca_system_score_gemma":0.0026155892,"threshold_uncertainty_score":0.020138025},"labels":[],"label_agreement":null},{"id":"W4389665609","doi":"10.1109/iros55552.2023.10341979","title":"Discovering Adaptable Symbolic Algorithms from Scratch","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Computer science; Scratch; Task (project management); Modular design; Robot; Swift; Artificial neural network; Control (management); Artificial intelligence; Inference; Algorithm; Programming language; Engineering","score_opus":0.025361716209531874,"score_gpt":0.2546971839649778,"score_spread":0.22933546775544594,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389665609","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11723005,0.0003219645,0.8737064,0.0003531106,0.00005544438,0.00010762904,0.00013685996,0.004593979,0.003494642],"genre_scores_gemma":[0.77995956,0.00016667912,0.21663536,0.00022319364,0.000024613792,0.00016011513,0.00036025338,0.0004553365,0.0020148708],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99944717,0.00012914294,0.000029972402,0.00019656071,0.00013218402,0.000064940425],"domain_scores_gemma":[0.9982907,0.0009423862,0.00012998909,0.00044855327,0.0001292357,0.00005917124],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009130712,0.0010529632,0.00087578397,0.00057016517,0.00041219022,0.00074832275,0.0014827797,0.0010571586,0.0023203162],"category_scores_gemma":[0.0060729124,0.00065150385,0.0007293476,0.00028311223,0.0014170111,0.001748816,0.0015083782,0.0017411611,0.00048389335],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016638414,0.000141727,0.0032493747,0.00023956163,0.000109078464,0.0001876305,0.00019486682,0.77782303,0.012441358,0.020719605,0.0019330529,0.18279438],"study_design_scores_gemma":[0.000016112006,0.000047596826,0.00015507913,0.000012243629,0.000010176892,0.000027415992,0.000022841998,0.9859519,0.0024632413,0.010653896,0.00063276687,0.0000066535904],"about_ca_topic_score_codex":0.0016653237,"about_ca_topic_score_gemma":0.0034209853,"teacher_disagreement_score":0.0023203162,"about_ca_system_score_codex":0.0007868831,"about_ca_system_score_gemma":0.0009880089,"threshold_uncertainty_score":0.0077622533},"labels":[],"label_agreement":null},{"id":"W4389667406","doi":"10.1109/iros55552.2023.10342408","title":"Dynamic Decision Frequency with Continuous Options","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Hyperparameter; Computer science; Reinforcement learning; Abstraction; Duration (music); Task (project management); Control (management); Action (physics); Variable (mathematics); Artificial intelligence; Mathematics; Engineering","score_opus":0.010684387379290488,"score_gpt":0.2540970301951204,"score_spread":0.24341264281582994,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389667406","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09374671,0.00044636763,0.8996291,0.0003425288,0.000094331386,0.00011607097,0.00005418701,0.00069221656,0.004878494],"genre_scores_gemma":[0.92952913,0.00010241253,0.06861135,0.000083897816,0.000021882712,0.00012280974,0.000036654514,0.000033457774,0.0014584907],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99928135,0.00023099213,0.00004288078,0.0001556159,0.00020516006,0.00008402554],"domain_scores_gemma":[0.99734986,0.0018085592,0.00024140571,0.00024262133,0.00017759006,0.00017993448],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013962394,0.0008031651,0.00072994427,0.0003691458,0.00036176245,0.0009720002,0.0013876706,0.0009283862,0.0028198708],"category_scores_gemma":[0.0055576595,0.00037856572,0.00042745718,0.00039022852,0.0012925232,0.0016321874,0.0012861406,0.0015768508,0.00028470843],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033703964,0.0001448926,0.0012886554,0.00010522269,0.00003498458,0.00011537578,0.00017569817,0.8936866,0.0028980772,0.0250378,0.00057144626,0.075604245],"study_design_scores_gemma":[0.000044049408,0.000068352965,0.00014095905,0.00001025604,0.0000075195967,0.000025603826,0.000011592217,0.9898858,0.00067969895,0.008507987,0.0006075463,0.000010618093],"about_ca_topic_score_codex":0.002635316,"about_ca_topic_score_gemma":0.0020553812,"teacher_disagreement_score":0.0028198708,"about_ca_system_score_codex":0.00082989206,"about_ca_system_score_gemma":0.0010185494,"threshold_uncertainty_score":0.009433389},"labels":[],"label_agreement":null},{"id":"W4390489156","doi":"10.1109/ssci52147.2023.10371909","title":"Hierarchical Reinforcement Learning for Non-Stationary Environments","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Carleton University","funders":"","keywords":"Reinforcement learning; Computer science; Process (computing); Temporal difference learning; Action (physics); Differential game; Term (time); Artificial intelligence; Mathematical optimization; Mathematics","score_opus":0.02224100655619201,"score_gpt":0.2667573711172034,"score_spread":0.24451636456101136,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390489156","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06730651,0.0002993218,0.9289917,0.00028080068,0.000038312497,0.00006283887,0.000037100326,0.0005647226,0.0024187057],"genre_scores_gemma":[0.9581595,0.00009202505,0.039595965,0.000080870755,0.000018885492,0.00007885979,0.00004773851,0.000028439683,0.0018978233],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99935037,0.000212386,0.000033559507,0.00014533916,0.00013857796,0.00011975904],"domain_scores_gemma":[0.99779105,0.0013831403,0.00028066672,0.00015045702,0.00023647063,0.00015819001],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013399931,0.0007290999,0.00090100325,0.00039108674,0.000382955,0.00060844375,0.0011183396,0.00082290056,0.0018304626],"category_scores_gemma":[0.0049455445,0.00034210682,0.00040199363,0.00023213551,0.001038216,0.0010205646,0.0011043709,0.0012755024,0.00023399299],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008085149,0.00006808189,0.00088154577,0.000052145922,0.000025785019,0.000079770056,0.00008692831,0.96285695,0.0013362595,0.010904854,0.00041787472,0.023209048],"study_design_scores_gemma":[0.000008986557,0.000022457216,0.000080393176,0.0000019166705,0.000002539338,0.0000046342907,0.000003998652,0.99560463,0.00014339494,0.00403127,0.00009340751,0.0000024913206],"about_ca_topic_score_codex":0.009505447,"about_ca_topic_score_gemma":0.007901312,"teacher_disagreement_score":0.009505447,"about_ca_system_score_codex":0.0011883695,"about_ca_system_score_gemma":0.0011944876,"threshold_uncertainty_score":0.018900275},"labels":[],"label_agreement":null},{"id":"W4390708456","doi":"10.1007/978-3-031-23161-2_18","title":"StarCraft Bots and Competitions","year":2024,"lang":"en","type":"book-chapter","venue":"Encyclopedia of Computer Graphics and Games","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science","score_opus":0.008624342207372153,"score_gpt":0.21474381820000343,"score_spread":0.20611947599263128,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390708456","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023305232,0.010681669,0.06907886,0.0033631371,0.0010900581,0.00007900942,0.00031519312,0.0007752504,0.8913115],"genre_scores_gemma":[0.39589292,0.008848108,0.025624424,0.0008514699,0.0003572588,0.00015405186,0.0006385546,0.00041220634,0.56722105],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9997434,0.000055379387,0.00000802965,0.0000481745,0.00008962187,0.000055454595],"domain_scores_gemma":[0.9997367,0.00009791123,0.000018254417,0.00002790587,0.000038943872,0.0000802947],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002852627,0.0004105483,0.00036916425,0.00060155836,0.00096281274,0.0021853487,0.00070797995,0.0009871855,0.032691147],"category_scores_gemma":[0.001121288,0.00027916304,0.00024461545,0.00073295797,0.001411992,0.002230645,0.0014094488,0.0012040696,0.0035312625],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000834879,0.000053712636,0.0003087991,0.00015633706,0.000015942896,0.000091309834,0.00035380985,0.016714554,0.000798535,0.73767823,0.084394455,0.1593509],"study_design_scores_gemma":[0.000021993124,0.000052369065,0.00073029695,0.0002067274,0.0000059777494,0.00022933015,0.00031857155,0.02655351,0.000538882,0.48494667,0.48637417,0.000021435539],"about_ca_topic_score_codex":0.0038657957,"about_ca_topic_score_gemma":0.00812753,"teacher_disagreement_score":0.032691147,"about_ca_system_score_codex":0.0014069021,"about_ca_system_score_gemma":0.0008027863,"threshold_uncertainty_score":0.10936278},"labels":[],"label_agreement":null},{"id":"W4390709078","doi":"10.1007/978-3-031-23161-2_524","title":"Trustworthy Embodied Virtual Agents","year":2024,"lang":"en","type":"book-chapter","venue":"Encyclopedia of Computer Graphics and Games","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Embodied cognition; Trustworthiness; Computer science; Embodied agent; Human–computer interaction; Internet privacy; Artificial intelligence","score_opus":0.012367909730939588,"score_gpt":0.22788966891602597,"score_spread":0.21552175918508637,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390709078","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023502031,0.004469576,0.6647238,0.0013189719,0.00071742584,0.000084670224,0.00011584665,0.0009366854,0.3041309],"genre_scores_gemma":[0.57874745,0.004368985,0.12662286,0.0002422054,0.00020838749,0.0002083845,0.00024365452,0.00029375716,0.28906438],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996055,0.000089666435,0.000019744837,0.000065419066,0.0001869679,0.000032762713],"domain_scores_gemma":[0.9995351,0.00020541226,0.00004418408,0.00010456859,0.00006584426,0.00004485861],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003737014,0.00061577634,0.00035083204,0.0003014268,0.00052918657,0.001779804,0.0007796431,0.001015242,0.0108304825],"category_scores_gemma":[0.001979424,0.00036145202,0.00021507037,0.00023127721,0.0011715445,0.0018753329,0.0021032828,0.0011714632,0.002312767],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011186452,0.000054060725,0.00024442302,0.00024340244,0.000029292438,0.00027773992,0.00070140185,0.058723226,0.010849428,0.7144727,0.013535808,0.20075658],"study_design_scores_gemma":[0.00003611469,0.0001333338,0.0004257183,0.0001987495,0.000028035416,0.00042474622,0.0003137055,0.206407,0.00924835,0.5831449,0.199598,0.000041379797],"about_ca_topic_score_codex":0.00063489616,"about_ca_topic_score_gemma":0.0008507165,"teacher_disagreement_score":0.0108304825,"about_ca_system_score_codex":0.0004959168,"about_ca_system_score_gemma":0.00046717768,"threshold_uncertainty_score":0.036231577},"labels":[],"label_agreement":null},{"id":"W4390722476","doi":"10.48550/arxiv.2401.03529","title":"Quantifying stability of non-power-seeking in artificial agents","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Shutdown; Computer science; Metric (unit); Stability (learning theory); Core (optical fiber); Shut down; Power (physics); Mathematical optimization; Economics; Engineering; Mathematics; Operations management; Machine learning","score_opus":0.1637298112787587,"score_gpt":0.24617750634206914,"score_spread":0.08244769506331043,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390722476","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25589103,0.00048188816,0.7360859,0.001518786,0.00004092225,0.00011788738,0.00015776623,0.00048204305,0.005223778],"genre_scores_gemma":[0.9577084,0.0001548864,0.040633496,0.0001227128,0.000027646674,0.00011246753,0.000096423886,0.00009669471,0.0010472899],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.994155,0.0022585613,0.00043499202,0.0012589564,0.001456383,0.00043618298],"domain_scores_gemma":[0.8979132,0.077052824,0.011676579,0.0062132594,0.004483449,0.0026605946],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0088075865,0.0011728967,0.0010659584,0.0020177998,0.0009476453,0.002441945,0.0020953852,0.0020340155,0.0025869936],"category_scores_gemma":[0.07574522,0.0008533177,0.001306839,0.0007794489,0.005859689,0.005587686,0.0038768032,0.0028554557,0.00029337563],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020385621,0.000085335945,0.00764832,0.0002218245,0.00021306406,0.00024808364,0.0008496326,0.72385645,0.0060646576,0.24447335,0.00040654006,0.015728844],"study_design_scores_gemma":[0.00001799269,0.00016308899,0.00068499584,0.000034687804,0.000031069936,0.0000890919,0.00008453034,0.8083546,0.0027732821,0.18706621,0.00066740275,0.000033067092],"about_ca_topic_score_codex":0.0020798123,"about_ca_topic_score_gemma":0.0007734472,"teacher_disagreement_score":0.0088075865,"about_ca_system_score_codex":0.0028991974,"about_ca_system_score_gemma":0.0014780423,"threshold_uncertainty_score":0.04657954},"labels":[],"label_agreement":null},{"id":"W4391021135","doi":"10.1109/cdc49753.2023.10383590","title":"Weighted-Norm Bounds on Model Approximation in MDPs with Unbounded Per-Step Cost","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"National Science Foundation","keywords":"Bounding overwatch; Norm (philosophy); Function (biology); Computer science; Algorithm; Discrete mathematics; Combinatorics; Mathematics; Artificial intelligence; Philosophy","score_opus":0.026072486934639563,"score_gpt":0.2568377678037991,"score_spread":0.2307652808691595,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391021135","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014707582,0.0015247148,0.9750762,0.0014954546,0.00013168028,0.00010860314,0.0003013815,0.00042995176,0.006224326],"genre_scores_gemma":[0.66954446,0.003004621,0.31422442,0.0009908468,0.00036206632,0.0010701456,0.0016820661,0.0006150917,0.008506186],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.993182,0.00298835,0.00032508175,0.0010684902,0.0016518126,0.00078418],"domain_scores_gemma":[0.9498321,0.04238505,0.0021387255,0.0018392238,0.0024779288,0.0013269697],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011927838,0.0040704375,0.0038781902,0.002178155,0.0013266817,0.003782691,0.003631003,0.0034655703,0.0049447753],"category_scores_gemma":[0.052319232,0.0014592995,0.0023321817,0.001962627,0.0037451219,0.006791804,0.0050457423,0.0065090223,0.00090424117],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018944002,0.000063657324,0.00038505814,0.0001926311,0.00007546118,0.000052778607,0.000054183984,0.9512744,0.00021861691,0.03673149,0.00088554464,0.009876733],"study_design_scores_gemma":[0.000009725315,0.000031587282,0.000037428963,0.000026499352,0.000009359089,0.0000073040624,0.000008920052,0.9736486,0.00013307459,0.02585181,0.00022907497,0.0000066138796],"about_ca_topic_score_codex":0.01072762,"about_ca_topic_score_gemma":0.0071729105,"teacher_disagreement_score":0.011927838,"about_ca_system_score_codex":0.005310389,"about_ca_system_score_gemma":0.004668274,"threshold_uncertainty_score":0.063081205},"labels":[],"label_agreement":null},{"id":"W4391021298","doi":"10.1109/cdc49753.2023.10383636","title":"Asymmetric Actor-Critic with Approximate Information State","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Representation (politics); Markov decision process; Partially observable Markov decision process; State (computer science); Artificial intelligence; Observable; Complete information; Mathematical optimization; Markov chain; Theoretical computer science; Machine learning; Markov process; Algorithm; Mathematical economics; Markov model; Mathematics","score_opus":0.011392651288327471,"score_gpt":0.22457048515747524,"score_spread":0.21317783386914776,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391021298","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038611196,0.00030686703,0.9549397,0.00046332914,0.000057267065,0.00009506878,0.000088674125,0.00066607853,0.0047717276],"genre_scores_gemma":[0.9458416,0.00014131136,0.051031742,0.00014261677,0.00003930007,0.00015357253,0.00012823181,0.00007273822,0.0024488508],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9983773,0.00056679733,0.00008937838,0.00032576386,0.0004465909,0.00019415458],"domain_scores_gemma":[0.99130344,0.0060137166,0.0009586207,0.0007183765,0.00066512375,0.0003407623],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029868635,0.0022031236,0.0018867268,0.0005275775,0.0005003473,0.0014895522,0.001994871,0.001588998,0.0026244174],"category_scores_gemma":[0.0119784,0.00064087106,0.00047305078,0.00047948986,0.0015449469,0.0020898608,0.0020750258,0.0031989082,0.0004687488],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000115213836,0.00003375108,0.00028582977,0.00004233483,0.000017251486,0.000040050672,0.000018696619,0.9822955,0.00042145702,0.008408085,0.00033468616,0.007987075],"study_design_scores_gemma":[0.0000067795554,0.000013725453,0.000019265053,0.0000027363162,0.0000022833817,0.000004002323,0.0000015158007,0.99718493,0.00016072157,0.0025438895,0.00005820472,0.0000019348745],"about_ca_topic_score_codex":0.0036381655,"about_ca_topic_score_gemma":0.0024174303,"teacher_disagreement_score":0.0036381655,"about_ca_system_score_codex":0.0018186874,"about_ca_system_score_gemma":0.0017770013,"threshold_uncertainty_score":0.015796244},"labels":[],"label_agreement":null},{"id":"W4391093351","doi":"10.1109/bigdata59044.2023.10386490","title":"GPT-in-the-Loop: Supporting Adaptation in Multiagent Systems","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Adaptation (eye); Computer science; Loop (graph theory); Distributed computing; Neuroscience; Mathematics; Biology","score_opus":0.051188804408088005,"score_gpt":0.29544630193895277,"score_spread":0.24425749753086476,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391093351","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022566443,0.000080044476,0.97331357,0.00022334675,0.00002204616,0.000062765175,0.000020599808,0.0012880131,0.002423091],"genre_scores_gemma":[0.8282204,0.0001069635,0.16981998,0.00017028044,0.000023442797,0.00015197606,0.00004881227,0.00015252795,0.001305771],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994973,0.00022969092,0.000022025159,0.00010299979,0.00009623397,0.000051674226],"domain_scores_gemma":[0.9978951,0.0014547276,0.00014490083,0.00028455374,0.00011627242,0.00010446834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013698179,0.00058508775,0.0005059376,0.00026424968,0.00038960404,0.0008625642,0.0017393404,0.0011999232,0.0026245597],"category_scores_gemma":[0.0059982226,0.0003223821,0.0005218253,0.00023251276,0.0014612833,0.0017964527,0.002184261,0.0017528129,0.00035135535],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007249502,0.00008794891,0.0012976096,0.00008210023,0.00004864417,0.00021718028,0.0004898274,0.8911677,0.004319448,0.032223612,0.00090988295,0.06908354],"study_design_scores_gemma":[0.000011548369,0.000025568039,0.000060253056,0.0000047828853,0.000005429952,0.000020494967,0.0000152893,0.9821584,0.0006245681,0.016262617,0.00080679107,0.000004381536],"about_ca_topic_score_codex":0.002392276,"about_ca_topic_score_gemma":0.0017272326,"teacher_disagreement_score":0.0026245597,"about_ca_system_score_codex":0.0005203846,"about_ca_system_score_gemma":0.00071763195,"threshold_uncertainty_score":0.008780062},"labels":[],"label_agreement":null},{"id":"W4391157052","doi":"","title":"Using Confounded Data in Latent Model-Based Reinforcement Learning","year":2023,"lang":"en","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ubisoft (Canada); Montfort Hospital; Minnow Environmental (Canada)","funders":"Canada First Research Excellence Fund; Canada Excellence Research Chairs, Government of Canada","keywords":"Reinforcement learning; Computer science; Reinforcement; Latent inhibition; Artificial intelligence; Cognitive psychology; Machine learning; Psychology; Statistics; Social psychology; Mathematics","score_opus":0.06579403889571468,"score_gpt":0.28468535235101183,"score_spread":0.21889131345529717,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391157052","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023783678,0.0003054987,0.97429514,0.0004466413,0.00006045844,0.00005273424,0.00008159465,0.00043292026,0.00054137333],"genre_scores_gemma":[0.877867,0.0002051851,0.11839731,0.00036550636,0.00010494943,0.00026047992,0.0002867302,0.00014663249,0.002366176],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99596006,0.0025830856,0.00017074627,0.00066339097,0.00038550733,0.00023728842],"domain_scores_gemma":[0.9488193,0.044501644,0.0018401715,0.002058072,0.0018493393,0.0009315074],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010174425,0.0013155788,0.0034189408,0.00076571613,0.0007446848,0.0018674611,0.0033679842,0.0033654352,0.003818823],"category_scores_gemma":[0.05324675,0.0015292262,0.0009079324,0.0010370476,0.0026567348,0.00412324,0.004157359,0.005068934,0.0006235921],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00060567004,0.00017847844,0.0015199252,0.00018635846,0.000114169234,0.00009382505,0.00012308171,0.93485844,0.00080722774,0.020825293,0.0009853635,0.0397022],"study_design_scores_gemma":[0.000018119094,0.000021090358,0.000051997416,0.000006021963,0.0000056720532,0.000004360206,0.000002397035,0.9927618,0.00010629944,0.006957294,0.00005985631,0.0000049603295],"about_ca_topic_score_codex":0.005903688,"about_ca_topic_score_gemma":0.0054356134,"teacher_disagreement_score":0.010174425,"about_ca_system_score_codex":0.001801305,"about_ca_system_score_gemma":0.0018641745,"threshold_uncertainty_score":0.053808153},"labels":[],"label_agreement":null},{"id":"W4391237386","doi":"10.1109/cvci59596.2023.10397147","title":"Continual Reinforcement Learning for Autonomous Driving with Application on Velocity Control under Various Environment","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Intelligent Mechatronic Systems (Canada); University of Waterloo","funders":"","keywords":"Reinforcement learning; Adaptability; Computer science; Forgetting; Task (project management); Artificial intelligence; Control engineering; Engineering","score_opus":0.01069531595903084,"score_gpt":0.22723260846090212,"score_spread":0.2165372925018713,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391237386","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049020533,0.0002999102,0.9472947,0.00019568401,0.000045356683,0.000042329448,0.000013001354,0.00033696086,0.002751648],"genre_scores_gemma":[0.97434855,0.00009519827,0.024291493,0.000030467585,0.000015108563,0.00004831237,0.000012113541,0.000012930336,0.0011458014],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997234,0.00007929123,0.000014311629,0.00006443455,0.00007952873,0.00003911647],"domain_scores_gemma":[0.9992931,0.00035286453,0.00008765268,0.000049285594,0.00016489497,0.000052208696],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00096641073,0.0005396242,0.00046738528,0.00030551638,0.0003323323,0.0004434562,0.00077625085,0.0004977453,0.0010441805],"category_scores_gemma":[0.0016805618,0.00021680871,0.00031897938,0.00018637515,0.0007537829,0.0004776596,0.00069640717,0.0008656283,0.000115634284],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006142597,0.000064624895,0.00087689667,0.000057227786,0.00002574013,0.000071104536,0.00007530116,0.9438455,0.003286397,0.0060715266,0.00031773408,0.045246493],"study_design_scores_gemma":[0.0000040139175,0.000026453967,0.00006606214,0.0000018773013,0.0000026540677,0.0000074097616,0.000002892791,0.99861956,0.00026983814,0.0008476446,0.00014899448,0.0000025656218],"about_ca_topic_score_codex":0.0043409886,"about_ca_topic_score_gemma":0.0034722541,"teacher_disagreement_score":0.0043409886,"about_ca_system_score_codex":0.0005786168,"about_ca_system_score_gemma":0.00083099824,"threshold_uncertainty_score":0.008631408},"labels":[],"label_agreement":null},{"id":"W4391494285","doi":"10.1007/s13042-023-02063-6","title":"Expected Lenient Q-learning: a fast variant of the Lenient Q-learning algorithm for cooperative stochastic Markov games","year":2024,"lang":"en","type":"article","venue":"International Journal of Machine Learning and Cybernetics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Rimouski","funders":"Centre National pour la Recherche Scientifique et Technique","keywords":"Algorithm; Convergence (economics); Reinforcement learning; Markov chain; Computer science; Q-learning; Rate of convergence; Artificial intelligence; Markov decision process; Mathematics; Markov process; Machine learning; Key (lock)","score_opus":0.007476805942429175,"score_gpt":0.25780049233344815,"score_spread":0.250323686391019,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391494285","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0041348804,0.00009452446,0.9939779,0.00014290569,0.000059484977,0.0000774598,0.000025005724,0.00046176207,0.0010260245],"genre_scores_gemma":[0.383566,0.00023009046,0.6061277,0.00063412526,0.00015905741,0.00067301275,0.00028292232,0.0004908016,0.007836296],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981311,0.0007823925,0.000097457596,0.00027788302,0.0004775473,0.00023369145],"domain_scores_gemma":[0.9923711,0.005172783,0.00032506516,0.0005930639,0.0011251643,0.00041281592],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005183722,0.001399465,0.0023440944,0.0010961964,0.00087368145,0.0015746759,0.0041963737,0.0023757932,0.008748724],"category_scores_gemma":[0.01558007,0.0008025006,0.0008165435,0.0010952558,0.0021514231,0.0026457983,0.0040028663,0.0036066775,0.0017016878],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004006005,0.00021936583,0.000982182,0.00015554758,0.00007675683,0.00011220877,0.00014835718,0.770627,0.0012040524,0.06613795,0.0045065065,0.15542942],"study_design_scores_gemma":[0.000021529728,0.00002253391,0.000028268842,0.00000531135,0.0000030448296,0.000009688166,0.0000037092004,0.9904461,0.00015426002,0.00904129,0.00025992718,0.000004402928],"about_ca_topic_score_codex":0.006754462,"about_ca_topic_score_gemma":0.0058782063,"teacher_disagreement_score":0.008748724,"about_ca_system_score_codex":0.0017856386,"about_ca_system_score_gemma":0.0034047056,"threshold_uncertainty_score":0.02926737},"labels":[],"label_agreement":null},{"id":"W4392123244","doi":"10.1007/978-3-031-52554-4_1","title":"Reinforcement Learning Background","year":2024,"lang":"en","type":"book-chapter","venue":"SpringerBriefs in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University; Memorial University of Newfoundland","funders":"","keywords":"Reinforcement learning; Reinforcement; Psychology; Computer science; Artificial intelligence; Social psychology","score_opus":0.026025942293179557,"score_gpt":0.25863785593703387,"score_spread":0.2326119136438543,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392123244","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0020631647,0.05107676,0.34943992,0.0057485434,0.0021910255,0.00008911022,0.0006414694,0.0010975508,0.58765244],"genre_scores_gemma":[0.15682119,0.07212338,0.14414588,0.003596487,0.004435873,0.0003707963,0.0013866267,0.0008102931,0.6163095],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9996563,0.0000633942,0.000014629805,0.000118168115,0.000115833594,0.000031617077],"domain_scores_gemma":[0.99937797,0.00033792492,0.000026716425,0.000075726246,0.00013196154,0.000049675225],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00044742675,0.0009862116,0.00072151737,0.00064872805,0.0005586362,0.002234779,0.0014883816,0.001391425,0.051432345],"category_scores_gemma":[0.0017432983,0.00039375422,0.0005206964,0.0012064714,0.00096449524,0.0021366186,0.00096720114,0.0026131012,0.017291067],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000025876256,0.00011694266,0.00013667354,0.0004829411,0.000016993821,0.00006679741,0.00008496732,0.017291332,0.000595331,0.6116753,0.06366051,0.30584633],"study_design_scores_gemma":[0.0000127091935,0.000047854795,0.00020160536,0.00028870345,0.000014778649,0.00012999712,0.000026912856,0.030740827,0.0007822186,0.45325345,0.5144795,0.000021364993],"about_ca_topic_score_codex":0.003089454,"about_ca_topic_score_gemma":0.0021777132,"teacher_disagreement_score":0.051432345,"about_ca_system_score_codex":0.0015757775,"about_ca_system_score_gemma":0.0011673272,"threshold_uncertainty_score":0.17205834},"labels":[],"label_agreement":null},{"id":"W4392240640","doi":"10.3390/math12050709","title":"Dimensionless Policies Based on the Buckingham π Theorem: Is This a Good Way to Generalize Numerical Results?","year":2024,"lang":"en","type":"article","venue":"Mathematics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Context (archaeology); Inverted pendulum; Scaling; Mathematics; Linear-quadratic regulator; Controller (irrigation); Control theory (sociology); Optimal control; Dimensionless quantity; Control (management); Mathematical optimization; Computer science; Applied mathematics; Artificial intelligence; Nonlinear system; Geometry; Physics","score_opus":0.030967814446170003,"score_gpt":0.27863134035175635,"score_spread":0.24766352590558635,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392240640","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0071800617,0.0005590678,0.97940814,0.0017337705,0.0004002607,0.000042284642,0.000039158516,0.000216565,0.010420758],"genre_scores_gemma":[0.49271327,0.002290462,0.49225542,0.002914072,0.00070615805,0.0005225022,0.00015951459,0.00037488542,0.008063602],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9988418,0.00041735652,0.00009735452,0.00025269738,0.0003292632,0.0000614959],"domain_scores_gemma":[0.9967907,0.0017517733,0.00036055513,0.0007858742,0.0002069868,0.0001041357],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023406034,0.0008582629,0.0010391908,0.00063247106,0.00061256514,0.001580393,0.0012940284,0.0014990139,0.0050393594],"category_scores_gemma":[0.013665888,0.0004385733,0.0010650845,0.00043094985,0.0041189934,0.0057459543,0.001975952,0.0034748637,0.0006941245],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000029596646,0.000023169741,0.0002586097,0.00011136313,0.000022382348,0.000045215573,0.000109286375,0.10901128,0.0011145149,0.86601853,0.0010451835,0.022210859],"study_design_scores_gemma":[0.000027311165,0.00009483206,0.00015707396,0.00006662197,0.000008312104,0.000042267693,0.000041363604,0.2668634,0.0013164992,0.72125596,0.010093483,0.00003275671],"about_ca_topic_score_codex":0.001175967,"about_ca_topic_score_gemma":0.0004096442,"teacher_disagreement_score":0.0050393594,"about_ca_system_score_codex":0.0009740312,"about_ca_system_score_gemma":0.0009268792,"threshold_uncertainty_score":0.01685834},"labels":[],"label_agreement":null},{"id":"W4392909880","doi":"10.1109/icassp48485.2024.10447501","title":"A Robust Quantile Huber Loss with Interpretable Parameter Adjustment in Distributional Reinforcement Learning","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Quantile; Reinforcement learning; Computer science; Artificial intelligence; Econometrics; Robustness (evolution); Reinforcement; Mathematical optimization; Mathematics; Machine learning; Statistics; Engineering","score_opus":0.01965315354986084,"score_gpt":0.24414673776042978,"score_spread":0.22449358421056895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392909880","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01227144,0.00020990161,0.9858678,0.00024278616,0.000030152,0.00003674977,0.00003398782,0.000322505,0.0009846879],"genre_scores_gemma":[0.84785736,0.0003573534,0.14642997,0.00042808495,0.00010457883,0.00021150743,0.00017599115,0.0002196353,0.0042155],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.998016,0.00087599707,0.000095547264,0.00042511747,0.00041602418,0.00017130675],"domain_scores_gemma":[0.99498993,0.0031181313,0.00051782664,0.0005307776,0.00062778505,0.00021548483],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004571722,0.0013099156,0.001536621,0.00064541027,0.0004554125,0.0014923953,0.002024994,0.0018547615,0.0023258696],"category_scores_gemma":[0.01752648,0.0004757568,0.0006151713,0.0006713965,0.0021193034,0.0027388742,0.0022126862,0.0026285718,0.00048173624],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013684943,0.000116156116,0.0012678654,0.000094532654,0.000059367267,0.00011401227,0.00010941462,0.8795429,0.0025991625,0.05180491,0.0018731564,0.06228155],"study_design_scores_gemma":[0.000010586463,0.000047870053,0.0001069814,0.000009182849,0.0000066021084,0.000017999371,0.000006348084,0.9845116,0.0004161917,0.014600203,0.0002569071,0.00000955255],"about_ca_topic_score_codex":0.0020098188,"about_ca_topic_score_gemma":0.0012826951,"teacher_disagreement_score":0.004571722,"about_ca_system_score_codex":0.0013515005,"about_ca_system_score_gemma":0.0015376828,"threshold_uncertainty_score":0.02417785},"labels":[],"label_agreement":null},{"id":"W4392964256","doi":"10.1007/978-3-031-56027-9_20","title":"A Streaming Approach to Neural Team Formation Training","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Computer science; Training (meteorology); Artificial intelligence; Artificial neural network; Multimedia","score_opus":0.03235137083435407,"score_gpt":0.2533138865639717,"score_spread":0.22096251572961761,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392964256","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0034949153,0.0002784097,0.99248946,0.00009891188,0.00006909147,0.000025745703,0.00005841798,0.00034288003,0.0031421229],"genre_scores_gemma":[0.2919327,0.0009440021,0.6868552,0.00026880324,0.00046757035,0.0003716915,0.0005731909,0.00041986207,0.018167004],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997466,0.00007326938,0.000015664755,0.000058554655,0.00007873245,0.000027105456],"domain_scores_gemma":[0.9989243,0.00065093697,0.000044341406,0.00016566669,0.00014512957,0.00006959569],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007511095,0.00066741527,0.00092721666,0.00038178838,0.00042397084,0.0006218585,0.0025714575,0.0011209532,0.01087198],"category_scores_gemma":[0.0025056314,0.0005281996,0.0004917437,0.0008504862,0.0006059475,0.0015915653,0.0017583115,0.001953612,0.0012895807],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001189607,0.00010477359,0.00022810619,0.00011746,0.000041335097,0.00005815734,0.00007738285,0.69886065,0.0026446113,0.06318925,0.009099218,0.22546016],"study_design_scores_gemma":[0.0000056045255,0.000014364576,0.00002222653,0.000003761481,0.0000024777887,0.00000880855,0.0000033707663,0.97850215,0.00017173703,0.020593964,0.0006695615,0.0000019363642],"about_ca_topic_score_codex":0.0026505901,"about_ca_topic_score_gemma":0.0033463482,"teacher_disagreement_score":0.01087198,"about_ca_system_score_codex":0.00062224537,"about_ca_system_score_gemma":0.0005195633,"threshold_uncertainty_score":0.036370397},"labels":[],"label_agreement":null},{"id":"W4393121880","doi":"10.1007/978-3-031-56855-8_22","title":"Using Evolution and Deep Learning to Generate Diverse Intelligent Agents","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Brock University","funders":"","keywords":"Computer science; Artificial intelligence; Deep learning","score_opus":0.04096502972223453,"score_gpt":0.2808853062363742,"score_spread":0.23992027651413966,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393121880","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12689073,0.0004914448,0.85599315,0.00053761125,0.00018457894,0.00012974582,0.00007177049,0.0008577462,0.014843236],"genre_scores_gemma":[0.71867555,0.00021338249,0.27298924,0.00021217087,0.00004231286,0.00018159483,0.00013580502,0.00018673707,0.00736334],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99980766,0.00004396068,0.000010898957,0.00004403327,0.000059587295,0.000033885146],"domain_scores_gemma":[0.9993735,0.00031314834,0.000051268675,0.00009314739,0.00011536058,0.00005368363],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00081084267,0.0006138854,0.00059315015,0.00056323764,0.0005569774,0.0007811684,0.0012478473,0.0011077324,0.002160265],"category_scores_gemma":[0.0020739944,0.00053527136,0.0005492737,0.00040542087,0.0011417989,0.0009330528,0.0016280814,0.001430662,0.0003584553],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000065120454,0.00009534905,0.0008146705,0.000060170307,0.000057378562,0.00009652361,0.00013266037,0.85001814,0.008005943,0.03862252,0.0013843303,0.1006472],"study_design_scores_gemma":[0.000013314378,0.000028351991,0.00006544335,0.0000070849223,0.000008036495,0.000017350756,0.000011907396,0.98447347,0.0013371041,0.01328268,0.00074991334,0.000005345374],"about_ca_topic_score_codex":0.0021571962,"about_ca_topic_score_gemma":0.003244068,"teacher_disagreement_score":0.002160265,"about_ca_system_score_codex":0.0010369384,"about_ca_system_score_gemma":0.0005046338,"threshold_uncertainty_score":0.0075235963},"labels":[],"label_agreement":null},{"id":"W4393146382","doi":"10.1609/aaai.v38i20.30613","title":"Reward-Respecting Subtasks for Model-Based Reinforcement Learning (Abstract Reprint)","year":2024,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Canadian Institute for Advanced Research","funders":"","keywords":"Reprint; Reinforcement learning; Reinforcement; Psychology; Computer science; Artificial intelligence; Social psychology; Physics","score_opus":0.10020869692937655,"score_gpt":0.3268577768747262,"score_spread":0.22664907994534966,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393146382","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0069100996,0.0002607348,0.9882515,0.00032637574,0.00007956726,0.00004110163,0.000079916164,0.00076662586,0.003284058],"genre_scores_gemma":[0.6709624,0.0004904431,0.3191465,0.00040932785,0.00014413969,0.00048689617,0.0004316566,0.00044364543,0.0074850037],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994881,0.00021510596,0.000028215467,0.000105703366,0.00009991677,0.000062789295],"domain_scores_gemma":[0.99884194,0.00073222833,0.00008983675,0.00014868427,0.00008796524,0.00009934931],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001165389,0.0011747809,0.00087149104,0.00030544618,0.00045714836,0.0010089619,0.001222663,0.001206673,0.007889611],"category_scores_gemma":[0.0041790996,0.000440855,0.000800336,0.00045303625,0.0013346296,0.0014038894,0.0018126155,0.0026851697,0.0012173946],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001496795,0.00008813665,0.00036412707,0.00009931136,0.00002936593,0.00010512022,0.00007057498,0.85790586,0.0016223332,0.08279026,0.0047202795,0.052055012],"study_design_scores_gemma":[0.00001125808,0.000017060063,0.000030613435,0.000008486909,0.0000035027067,0.000006872921,0.000002318462,0.95333445,0.00025070464,0.04565765,0.000672573,0.0000044854937],"about_ca_topic_score_codex":0.004716786,"about_ca_topic_score_gemma":0.0045567867,"teacher_disagreement_score":0.007889611,"about_ca_system_score_codex":0.0013382483,"about_ca_system_score_gemma":0.0012075728,"threshold_uncertainty_score":0.026393354},"labels":[],"label_agreement":null},{"id":"W4393157304","doi":"10.1609/aaai.v38i20.30606","title":"Exploiting Action Impact Regularity and Exogenous State Variables for Offline Reinforcement Learning (Abstract Reprint)","year":2024,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reprint; Reinforcement learning; Action (physics); Reinforcement; State action; Computer science; Cognitive psychology; Psychology; Artificial intelligence; Social psychology; Political science; Law; Physics","score_opus":0.11101470799603334,"score_gpt":0.3391604714125966,"score_spread":0.22814576341656323,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393157304","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016138358,0.000085998785,0.97990865,0.000286079,0.00005151566,0.000046824287,0.00005609716,0.00052527484,0.002901247],"genre_scores_gemma":[0.81941855,0.00014156457,0.17575683,0.00023784737,0.0000666218,0.00015984177,0.00015860847,0.0001850362,0.0038750204],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994967,0.00017413699,0.000030041025,0.00012041038,0.000115436545,0.0000632592],"domain_scores_gemma":[0.9956648,0.0029141002,0.00032235237,0.0006998408,0.00022065233,0.00017830227],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015928895,0.00058893627,0.00070368976,0.00028643376,0.00035029327,0.00074935466,0.00081519934,0.00070494495,0.0060725124],"category_scores_gemma":[0.008942412,0.0003453712,0.0003685837,0.00039248736,0.0014020762,0.0014261527,0.0013794545,0.0023154996,0.00077326846],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018694249,0.00010154576,0.0011224693,0.00006399741,0.00002227824,0.00010273785,0.00006200623,0.88167983,0.0023050802,0.039129052,0.002172824,0.07305124],"study_design_scores_gemma":[0.000013017116,0.000029169374,0.000098841854,0.0000049383466,0.0000021318588,0.000012627804,0.000003361201,0.98592293,0.0006486625,0.012733348,0.0005277365,0.000003163361],"about_ca_topic_score_codex":0.003364386,"about_ca_topic_score_gemma":0.0033778497,"teacher_disagreement_score":0.0060725124,"about_ca_system_score_codex":0.0006798785,"about_ca_system_score_gemma":0.0012974135,"threshold_uncertainty_score":0.020314634},"labels":[],"label_agreement":null},{"id":"W4393159809","doi":"10.1609/aaai.v38i10.28970","title":"Contextual Pre-planning on Reward Machine Abstractions for Enhanced Transfer in Deep Reinforcement Learning","year":2024,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Reinforcement; Artificial intelligence; Transfer of learning; Cognitive psychology; Machine learning; Psychology; Social psychology","score_opus":0.06539929613530107,"score_gpt":0.3236612521517399,"score_spread":0.25826195601643887,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393159809","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07160936,0.00024845547,0.92469925,0.00029127309,0.000040193365,0.000059304522,0.000070394366,0.0013113372,0.0016704574],"genre_scores_gemma":[0.92444396,0.00010291487,0.07359917,0.00010355096,0.000018829278,0.000116516785,0.00009991446,0.00007900817,0.0014360793],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996177,0.00014540686,0.000019399436,0.00008865491,0.000069894115,0.0000588063],"domain_scores_gemma":[0.9983728,0.0010324921,0.00015027192,0.0002309935,0.000107852466,0.000105495434],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010456064,0.0007436522,0.00074487925,0.0002718035,0.00030711995,0.0006107449,0.0011285665,0.0007602126,0.0025340575],"category_scores_gemma":[0.004879512,0.00042061973,0.00037562707,0.00027224104,0.0011258379,0.0014885234,0.0016025619,0.0020098917,0.00033984517],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000116296156,0.00010322752,0.00063566194,0.000053527077,0.000021159904,0.00006029937,0.00010340389,0.928771,0.002459962,0.012360986,0.0008023371,0.05451202],"study_design_scores_gemma":[0.000007910686,0.000029729435,0.000049144597,0.0000039160886,0.0000025416286,0.000004424575,0.000004601676,0.9910693,0.0005283452,0.008136575,0.00016057717,0.0000028224074],"about_ca_topic_score_codex":0.00306074,"about_ca_topic_score_gemma":0.0036897191,"teacher_disagreement_score":0.00306074,"about_ca_system_score_codex":0.0009782019,"about_ca_system_score_gemma":0.0012166,"threshold_uncertainty_score":0.008477271},"labels":[],"label_agreement":null},{"id":"W4393160751","doi":"10.1609/aaai.v38i18.30062","title":"Parallel Beam Search Algorithms for Domain-Independent Dynamic Programming","year":2024,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Dynamic programming; Computer science; Domain (mathematical analysis); Algorithm; Parallel computing; Mathematics","score_opus":0.07861063744368464,"score_gpt":0.33554137191464556,"score_spread":0.25693073447096093,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393160751","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003269799,0.00020432229,0.99311876,0.00012424203,0.000037297745,0.000047793023,0.000060676168,0.0006908625,0.002446129],"genre_scores_gemma":[0.120892905,0.00033603248,0.87406397,0.00027209264,0.000055643715,0.0005458795,0.0003409722,0.00043158518,0.0030609765],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988562,0.00040152908,0.00006630998,0.00021389572,0.0002996255,0.00016245876],"domain_scores_gemma":[0.99824595,0.0010209361,0.00011387708,0.00023437366,0.00031015373,0.000074793534],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002204216,0.0012644443,0.0014118219,0.0010567941,0.00087043,0.0014308385,0.002024596,0.001359553,0.00700154],"category_scores_gemma":[0.0044819163,0.0008718013,0.0013284499,0.0017206135,0.0010463394,0.0020541511,0.0021689609,0.00227054,0.0013834658],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016965164,0.00012569432,0.00066605257,0.00015068911,0.00011937622,0.00006316547,0.00011209654,0.7653571,0.0020383957,0.080914915,0.00857579,0.14170712],"study_design_scores_gemma":[0.00004334516,0.000016465277,0.000033629323,0.000007448717,0.000008004952,0.000010572849,0.00000967093,0.9768059,0.0004986669,0.020749923,0.0018099358,0.000006352381],"about_ca_topic_score_codex":0.0048140204,"about_ca_topic_score_gemma":0.006053125,"teacher_disagreement_score":0.00700154,"about_ca_system_score_codex":0.0011276916,"about_ca_system_score_gemma":0.0024339096,"threshold_uncertainty_score":0.02342254},"labels":[],"label_agreement":null},{"id":"W4393226450","doi":"10.1017/jfm.2023.1096","title":"Learn to flap: foil non-parametric path planning via deep reinforcement learning","year":2024,"lang":"en","type":"article","venue":"Journal of Fluid Mechanics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Reinforcement learning; Computer science; Path (computing); Parametric statistics; Reinforcement; FOIL method; Motion planning; Artificial intelligence; Materials science; Robot; Mathematics; Computer network","score_opus":0.015150885779447175,"score_gpt":0.2633649266548724,"score_spread":0.2482140408754252,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393226450","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.042469475,0.00029394453,0.9526489,0.00022485426,0.00005883916,0.000055617606,0.000047697413,0.0009418909,0.00325872],"genre_scores_gemma":[0.8848344,0.00012520741,0.11188682,0.00019286638,0.000022094016,0.00011191958,0.00012270585,0.00008449454,0.0026194616],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99984896,0.000035007397,0.0000074547283,0.00004212008,0.000035334462,0.00003113747],"domain_scores_gemma":[0.9995752,0.00021952776,0.000057907317,0.000037889185,0.00006761128,0.00004192339],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00051447557,0.0007159973,0.00051530177,0.00022961046,0.00017475616,0.00038218332,0.0007035157,0.0006727941,0.0020856427],"category_scores_gemma":[0.0016875772,0.00029136587,0.00026922903,0.00017793066,0.00055372575,0.0005854783,0.000756519,0.0008938607,0.00036614423],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000070631366,0.000047826776,0.00080392044,0.000051600506,0.000016842034,0.000064969834,0.00003846843,0.9388943,0.0028383015,0.0033075337,0.0011317235,0.05273393],"study_design_scores_gemma":[0.0000054931384,0.000024458286,0.000033566124,0.0000028058403,0.0000017204459,0.0000068005897,0.0000030839883,0.99855274,0.00036133925,0.0008325157,0.00017368028,0.00000186084],"about_ca_topic_score_codex":0.002715063,"about_ca_topic_score_gemma":0.002684429,"teacher_disagreement_score":0.002715063,"about_ca_system_score_codex":0.00043373276,"about_ca_system_score_gemma":0.00092345336,"threshold_uncertainty_score":0.0069772005},"labels":[],"label_agreement":null},{"id":"W4394708838","doi":"10.48550/arxiv.2404.05870","title":"CoBT: Collaborative Programming of Behaviour Trees from One Demonstration for Robot Manipulation","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"European Regional Development Fund; Science Foundation Ireland; European Commission; ADAPT - Centre for Digital Content Technology; Canadian Institute of Steel Construction","keywords":"Computer science; Robot; Human–computer interaction; Artificial intelligence","score_opus":0.09148361143653079,"score_gpt":0.220424809671624,"score_spread":0.1289411982350932,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394708838","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008982783,0.000045035595,0.9816139,0.00006439005,0.000019217623,0.00012282218,0.000074814096,0.007742313,0.0013348991],"genre_scores_gemma":[0.29404616,0.00009765867,0.6996199,0.00010154101,0.000014479097,0.00077015185,0.00044863822,0.0015899008,0.0033116553],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994288,0.00017572229,0.000030138784,0.00015044837,0.00016024878,0.00005459074],"domain_scores_gemma":[0.9981184,0.0010976918,0.00013657406,0.00034309516,0.00016067347,0.00014356326],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010970926,0.00091495056,0.00040961453,0.00039484765,0.00035520268,0.000527486,0.0020257179,0.0009960481,0.009069506],"category_scores_gemma":[0.0045269337,0.00048372362,0.00074490265,0.00020200458,0.00085186795,0.0008924179,0.0018530637,0.0012906262,0.0015303158],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005253384,0.00081103767,0.0031303556,0.00071100454,0.00013112623,0.00089209975,0.0012156809,0.37608796,0.09202178,0.030031292,0.012890484,0.48155183],"study_design_scores_gemma":[0.00006273496,0.00016845937,0.00049497193,0.000032095188,0.000015371752,0.00018346794,0.000047628346,0.9600698,0.016829737,0.012847892,0.009223522,0.000024366347],"about_ca_topic_score_codex":0.001321318,"about_ca_topic_score_gemma":0.0023840815,"teacher_disagreement_score":0.009069506,"about_ca_system_score_codex":0.0003686543,"about_ca_system_score_gemma":0.00070642785,"threshold_uncertainty_score":0.030340493},"labels":[],"label_agreement":null},{"id":"W4395703972","doi":"10.2139/ssrn.4771123","title":"Autonomous Robot Navigation in Dynamic Environments: A Temporal-Difference Learning Approach","year":2024,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Temporal difference learning; Computer science; Robot; Artificial intelligence; Human–computer interaction; Reinforcement learning","score_opus":0.007485515225629387,"score_gpt":0.23333869961515769,"score_spread":0.2258531843895283,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4395703972","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015146335,0.00019785482,0.98300296,0.00015477964,0.000046320536,0.000013367746,0.000013671708,0.000052717904,0.0013719958],"genre_scores_gemma":[0.8638048,0.0003035803,0.13158959,0.0001340868,0.00008646956,0.0000992172,0.000056948156,0.00005253259,0.0038727792],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996654,0.00010013015,0.000019229663,0.00008414702,0.00009225953,0.000038871927],"domain_scores_gemma":[0.9987722,0.00079005264,0.00010299194,0.00006314573,0.00017901711,0.0000925106],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013123145,0.000412473,0.00094803405,0.0003912525,0.00036478025,0.0006493887,0.0020281428,0.0010503859,0.001974709],"category_scores_gemma":[0.002595275,0.00040410936,0.0005518377,0.0005319359,0.0010594301,0.0013393883,0.0013355,0.0012310429,0.00017257716],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014875643,0.00011680805,0.00071142084,0.0000793081,0.000073202646,0.000052740168,0.000072489456,0.88228726,0.0023740798,0.049340382,0.0005383631,0.06420524],"study_design_scores_gemma":[0.000005766437,0.000021716798,0.000038868493,0.0000016829979,0.0000039092156,0.000005964532,0.0000019177776,0.99354494,0.00011222694,0.006170725,0.000089569025,0.0000027081314],"about_ca_topic_score_codex":0.0039975527,"about_ca_topic_score_gemma":0.0027109976,"teacher_disagreement_score":0.0039975527,"about_ca_system_score_codex":0.0007805739,"about_ca_system_score_gemma":0.00097505597,"threshold_uncertainty_score":0.007948577},"labels":[],"label_agreement":null},{"id":"W4396515242","doi":"10.22215/etd/2024-15896","title":"A Machine Learning Approach for Aerial Drones Playing the Pursuit-Evasion Game","year":2024,"lang":"en","type":"dissertation","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Drone; Reinforcement learning; Popularity; Computer science; Artificial intelligence; Pursuit-evasion; Machine learning; Fuzzy logic; Range (aeronautics); Engineering; Aerospace engineering; Psychology","score_opus":0.024853440727771136,"score_gpt":0.2744335396824072,"score_spread":0.24958009895463606,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396515242","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026458295,0.0003844867,0.9646883,0.0005183022,0.000048746482,0.000055560016,0.00001979184,0.000110399495,0.007716206],"genre_scores_gemma":[0.84677535,0.00041591705,0.14206347,0.0001283446,0.000066025386,0.00023642165,0.000039929993,0.000026567446,0.010247955],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998129,0.000059405527,0.00000761501,0.0000487129,0.000048867238,0.000022557086],"domain_scores_gemma":[0.9995254,0.00030914237,0.000040888342,0.00001424131,0.000086784865,0.000023531578],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005115726,0.0004938184,0.00058855634,0.00031838633,0.00041778822,0.00062072044,0.0008217076,0.00085514947,0.0024542501],"category_scores_gemma":[0.0015391038,0.00024722132,0.00047372223,0.00022292123,0.00067293877,0.00056987285,0.000567828,0.0010550182,0.00022832633],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000022903203,0.000043268927,0.0004528557,0.000044898567,0.000020713984,0.000052634503,0.00006370465,0.93603885,0.00088111166,0.03285612,0.0005011807,0.029021736],"study_design_scores_gemma":[0.0000022441534,0.000008003016,0.00002895341,0.00000209691,0.0000013333373,0.0000037650834,0.0000029694977,0.9970565,0.000057566842,0.0026950513,0.00014027691,0.0000013354446],"about_ca_topic_score_codex":0.0077057327,"about_ca_topic_score_gemma":0.0051415158,"teacher_disagreement_score":0.0077057327,"about_ca_system_score_codex":0.0011844176,"about_ca_system_score_gemma":0.000960621,"threshold_uncertainty_score":0.015321732},"labels":[],"label_agreement":null},{"id":"W4396534955","doi":"10.1109/tvt.2024.3394350","title":"You Only Look at Once for Real-Time and Generic Multi-Task","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Vehicular Technology","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":62,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"Canada Research Chairs","keywords":"Task (project management); Computer science; Engineering; Embedded system; Systems engineering","score_opus":0.014429421417233796,"score_gpt":0.25014116532074776,"score_spread":0.23571174390351396,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396534955","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.047441546,0.00086711266,0.91816854,0.0014518669,0.0004776323,0.00025335295,0.0009873123,0.01558001,0.014772735],"genre_scores_gemma":[0.66605526,0.00074989756,0.30213964,0.0010361166,0.00014790698,0.0003577363,0.0026993274,0.001239032,0.025575152],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.999617,0.000042473905,0.000018408284,0.00016141223,0.00008610384,0.00007462234],"domain_scores_gemma":[0.9993542,0.00012896785,0.000040164476,0.0002493345,0.00014404413,0.00008328712],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00064019253,0.0011688826,0.0006407889,0.0002521865,0.00046234502,0.0011077034,0.0019533248,0.0013597485,0.00652152],"category_scores_gemma":[0.0020227453,0.0005541351,0.00074383226,0.00034027474,0.00052960997,0.0027745797,0.0015023084,0.0022376645,0.005772929],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001168134,0.00081729057,0.007867209,0.00061917753,0.0002230504,0.00049205445,0.00038804216,0.27469313,0.05325721,0.018202506,0.08601648,0.5562558],"study_design_scores_gemma":[0.00004799335,0.00018384735,0.0013680871,0.000033241296,0.000035869987,0.0002081738,0.00008114383,0.94384336,0.011014989,0.014556892,0.028576912,0.000049424303],"about_ca_topic_score_codex":0.005415718,"about_ca_topic_score_gemma":0.010284913,"teacher_disagreement_score":0.00652152,"about_ca_system_score_codex":0.0005765337,"about_ca_system_score_gemma":0.0011709995,"threshold_uncertainty_score":0.02181667},"labels":[],"label_agreement":null},{"id":"W4397048993","doi":"10.48550/arxiv.2405.09999","title":"Reward Centering","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; DeepMind; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Psychology","score_opus":0.07759720201491045,"score_gpt":0.18758010996262264,"score_spread":0.10998290794771219,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4397048993","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013554957,0.0005344027,0.9798356,0.00024999748,0.00012758018,0.000074769494,0.000041152784,0.0009142033,0.004667346],"genre_scores_gemma":[0.5776055,0.0005321493,0.414327,0.00035398515,0.00015584522,0.00024235973,0.0001643235,0.00039543788,0.006223497],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.998064,0.0006061087,0.0000922085,0.0005319995,0.0005082643,0.00019740322],"domain_scores_gemma":[0.9947253,0.0025606768,0.00058684335,0.0009384144,0.0008606769,0.0003281211],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028533847,0.0012649093,0.001404197,0.00064550404,0.0006382671,0.0014369518,0.0022719535,0.00112007,0.0063287085],"category_scores_gemma":[0.014188872,0.00040894188,0.0006169833,0.00056888215,0.0016686599,0.0026835776,0.002309038,0.002513067,0.0012159749],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040368558,0.00033207313,0.0025099588,0.00034136156,0.00013284007,0.000106196385,0.00022075573,0.53126645,0.006059004,0.13397905,0.006795191,0.31785342],"study_design_scores_gemma":[0.000040203526,0.00012915605,0.0002525918,0.00003798615,0.00003095054,0.000055556262,0.000013568283,0.9523626,0.00404628,0.039076388,0.0039323187,0.000022397684],"about_ca_topic_score_codex":0.0024126938,"about_ca_topic_score_gemma":0.0019791753,"teacher_disagreement_score":0.0063287085,"about_ca_system_score_codex":0.0016945936,"about_ca_system_score_gemma":0.0019586158,"threshold_uncertainty_score":0.02117163},"labels":[],"label_agreement":null},{"id":"W4398186376","doi":"10.1145/3605098.3635992","title":"Explainable Artificial Intelligence (XAI) Approach for Reinforcement Learning Systems","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence","score_opus":0.051292487425204716,"score_gpt":0.28274909395273246,"score_spread":0.23145660652752775,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4398186376","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010683634,0.00017672115,0.9857848,0.0003630203,0.000032437612,0.00004797902,0.00005089695,0.00044865263,0.0024118063],"genre_scores_gemma":[0.7100282,0.0002653577,0.2867161,0.00018637159,0.000050447685,0.00023250014,0.00010411386,0.000087012704,0.002329837],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9993825,0.0002699928,0.000033172317,0.00010992094,0.00015726914,0.000047206577],"domain_scores_gemma":[0.9984549,0.0009861474,0.00017150767,0.00015970986,0.00017066623,0.00005703876],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013662486,0.00069734425,0.0005322803,0.00036426002,0.00034357977,0.0009212098,0.0012781073,0.0007904646,0.0027236626],"category_scores_gemma":[0.0046371557,0.00030075767,0.00055883033,0.00028931777,0.0009131282,0.00115317,0.001265683,0.0015688011,0.00024168058],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000054124353,0.000040095954,0.00097085803,0.00010690367,0.00005854152,0.0000898827,0.0001531345,0.88286906,0.0014108602,0.06224917,0.0008213354,0.05117611],"study_design_scores_gemma":[0.0000071469867,0.000021230499,0.000080571175,0.0000068082136,0.0000064931105,0.000014724587,0.0000071396566,0.97530293,0.0003570385,0.023384698,0.0008062198,0.000004881571],"about_ca_topic_score_codex":0.0034110085,"about_ca_topic_score_gemma":0.0030591425,"teacher_disagreement_score":0.0034110085,"about_ca_system_score_codex":0.0012248859,"about_ca_system_score_gemma":0.00088278385,"threshold_uncertainty_score":0.009111583},"labels":[],"label_agreement":null},{"id":"W4398850169","doi":"10.48550/arxiv.2405.14664","title":"Fisher Flow Matching for Generative Modeling over Discrete Data","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Engineering and Physical Sciences Research Council; Center for Evolutionary and Theoretical Immunology","keywords":"Matching (statistics); Flow (mathematics); Computer science; Generative grammar; Generative model; Mathematics; Econometrics; Algorithm; Artificial intelligence; Statistics; Geometry","score_opus":0.15335958160201563,"score_gpt":0.23882465678278314,"score_spread":0.08546507518076751,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4398850169","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005619896,0.00019784457,0.99270225,0.00029613185,0.00002885761,0.000031080097,0.00011522664,0.00035507756,0.0006536562],"genre_scores_gemma":[0.51720387,0.0010698539,0.46841574,0.00088434713,0.00029240036,0.00057227246,0.0016773896,0.0007264756,0.009157769],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99886084,0.0003929337,0.000058062782,0.00033478084,0.00026128715,0.00009209204],"domain_scores_gemma":[0.9949051,0.0037505378,0.000362707,0.000434614,0.00034940173,0.00019765094],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040117847,0.0011607552,0.0012824123,0.001860418,0.00083755195,0.0015331805,0.0019498847,0.0023461406,0.0041429293],"category_scores_gemma":[0.01417805,0.00089248887,0.0018937205,0.0012288643,0.0020628574,0.0029438932,0.0025883177,0.0031525688,0.0010157548],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006439642,0.000049548133,0.0012606874,0.00008641224,0.000039617644,0.00007485221,0.00011368873,0.8341947,0.0011321645,0.11035601,0.0019273144,0.050700527],"study_design_scores_gemma":[0.0000031258214,0.000006956519,0.000054575143,0.0000057586385,0.0000020682262,0.000008039407,0.0000035884575,0.9700714,0.00016737271,0.029282331,0.00038929423,0.0000054359293],"about_ca_topic_score_codex":0.009592979,"about_ca_topic_score_gemma":0.007267018,"teacher_disagreement_score":0.009592979,"about_ca_system_score_codex":0.0026935122,"about_ca_system_score_gemma":0.0019315942,"threshold_uncertainty_score":0.021216571},"labels":[],"label_agreement":null},{"id":"W4399019070","doi":"10.1016/j.eswa.2024.124310","title":"Bridging the simulation-to-real gap of depth images for deep reinforcement learning","year":2024,"lang":"en","type":"article","venue":"Expert Systems with Applications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Encoder; Bridging (networking); Bridge (graph theory); Perception; Virtual reality; Machine learning; Deep learning; Computer vision","score_opus":0.025106323295264334,"score_gpt":0.3069211344319391,"score_spread":0.2818148111366747,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399019070","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.051116355,0.00046263935,0.9436867,0.0008103324,0.000092290975,0.000032974494,0.000047002322,0.00045258566,0.003299057],"genre_scores_gemma":[0.94141537,0.00016762178,0.056956466,0.00016302831,0.000024730201,0.000046074983,0.00004724343,0.000079101555,0.0011003913],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99954814,0.00016372031,0.000022228101,0.000091823094,0.00012139621,0.000052594318],"domain_scores_gemma":[0.99642867,0.0025645052,0.00022568431,0.00034115114,0.00027510276,0.00016486744],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014539266,0.0006751512,0.00075736875,0.0002484349,0.00033856928,0.0009371083,0.001226212,0.0012455021,0.0033216607],"category_scores_gemma":[0.010221165,0.00050606014,0.0002614452,0.00018077712,0.0012195314,0.002088953,0.0023542014,0.0024654532,0.00026464855],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035999026,0.00013369323,0.00090141484,0.00016091272,0.00003647476,0.000094675306,0.00019204973,0.87780255,0.0063709626,0.03383234,0.0014137293,0.07870112],"study_design_scores_gemma":[0.000007868801,0.000028378014,0.000058804762,0.000007906125,0.0000019078761,0.0000069788075,0.00000718037,0.9897726,0.0007233365,0.009148688,0.0002336772,0.000002754359],"about_ca_topic_score_codex":0.0029550185,"about_ca_topic_score_gemma":0.0026714632,"teacher_disagreement_score":0.0033216607,"about_ca_system_score_codex":0.0010048758,"about_ca_system_score_gemma":0.0011977861,"threshold_uncertainty_score":0.011112034},"labels":[],"label_agreement":null},{"id":"W4399116101","doi":"10.48550/arxiv.2405.16899","title":"Partial Models for Building Adaptive Model-Based Reinforcement Learning Agents","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Reinforcement; Computer science; Artificial intelligence; Psychology; Social psychology","score_opus":0.12975458227712403,"score_gpt":0.22997985085964243,"score_spread":0.1002252685825184,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399116101","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021881575,0.00018546091,0.97483826,0.0003126096,0.000032740074,0.000051403385,0.00009373522,0.0007974845,0.0018066551],"genre_scores_gemma":[0.8016121,0.00027009263,0.19354238,0.00029633506,0.000037973954,0.00040198077,0.0002632388,0.00017305539,0.0034030075],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994836,0.00018493901,0.000034721652,0.00011098569,0.00012695612,0.00005877582],"domain_scores_gemma":[0.9984742,0.0007771968,0.00016221181,0.00024650845,0.00020823466,0.00013169754],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010014871,0.00094344665,0.0011452035,0.00045442398,0.0004428787,0.001138904,0.0016070626,0.001261079,0.002540833],"category_scores_gemma":[0.0046953782,0.00087858306,0.00097311346,0.00031849684,0.0014532503,0.0018921538,0.0023067016,0.0020916527,0.00053468265],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000026829726,0.000014275262,0.0003188573,0.00002922099,0.000027523058,0.00003093872,0.000055369044,0.9740397,0.0006091882,0.016600128,0.0003243725,0.007923584],"study_design_scores_gemma":[0.0000042938495,0.000013115292,0.000017877208,0.000003437157,0.000003979369,0.000005176176,0.0000034925197,0.9901932,0.00015167595,0.0093222605,0.00027852817,0.00000288766],"about_ca_topic_score_codex":0.004754286,"about_ca_topic_score_gemma":0.0064399997,"teacher_disagreement_score":0.004754286,"about_ca_system_score_codex":0.0011667663,"about_ca_system_score_gemma":0.0012441208,"threshold_uncertainty_score":0.009453177},"labels":[],"label_agreement":null},{"id":"W4399284119","doi":"10.1016/j.neucom.2024.127963","title":"Multi-fidelity reinforcement learning with control variates","year":2024,"lang":"en","type":"article","venue":"Neurocomputing","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Microsoft (Canada)","funders":"Argonne National Laboratory; Office of Science; Laboratory Computing Resource Center; Advanced Scientific Computing Research; U.S. Department of Energy","keywords":"Reinforcement learning; Variance reduction; Estimator; Fidelity; Control variates; Computer science; Variance (accounting); High fidelity; Mathematical optimization; Reduction (mathematics); Function (biology); Artificial intelligence; Machine learning; Monte Carlo method; Mathematics; Statistics","score_opus":0.014987009506667814,"score_gpt":0.24923637087076794,"score_spread":0.23424936136410013,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399284119","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024661936,0.0002065266,0.972298,0.00023868041,0.000056728233,0.000050998595,0.00002325007,0.00022508354,0.0022387423],"genre_scores_gemma":[0.9389747,0.000079840196,0.05819871,0.00011251372,0.000035467343,0.00013464446,0.000030796084,0.00004055012,0.0023927526],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988637,0.00047167024,0.00006397361,0.00018530755,0.00029009057,0.0001253061],"domain_scores_gemma":[0.99601245,0.0027654371,0.0003544539,0.00028283647,0.00043147552,0.0001533574],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002963822,0.00091214513,0.0010165026,0.0004854912,0.00039933508,0.00089655357,0.001325463,0.0018376277,0.0030593406],"category_scores_gemma":[0.008792929,0.0006098935,0.000529113,0.00042690278,0.0013000817,0.0014793596,0.0024419727,0.002028057,0.00029565254],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011771401,0.00006538557,0.00041267654,0.00005325293,0.000036654485,0.000050608953,0.000033899985,0.96752775,0.0010851477,0.008017759,0.0003130284,0.022286069],"study_design_scores_gemma":[0.000010804929,0.000035606674,0.00005516263,0.0000046197897,0.000004155446,0.000010769149,0.0000016943567,0.99789906,0.00027720653,0.0015989299,0.00009832414,0.000003530871],"about_ca_topic_score_codex":0.0026168497,"about_ca_topic_score_gemma":0.002348041,"teacher_disagreement_score":0.0030593406,"about_ca_system_score_codex":0.0007625231,"about_ca_system_score_gemma":0.0006987506,"threshold_uncertainty_score":0.015674412},"labels":[],"label_agreement":null},{"id":"W4399376477","doi":"10.1109/iciprob62548.2024.10543232","title":"Proximity-Based Reward Systems for Multi-Agent Reinforcement Learning","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Moncton","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Computer science; Swarm behaviour; Artificial intelligence; Swarm intelligence; Swarm robotics; Machine learning; Euclidean distance; Robot; Particle swarm optimization","score_opus":0.05601974589692608,"score_gpt":0.29988563588308553,"score_spread":0.24386588998615946,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399376477","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03172497,0.0004811333,0.9648021,0.00023436606,0.00006245183,0.00011446428,0.000031083076,0.0003854312,0.00216407],"genre_scores_gemma":[0.87708163,0.00018432703,0.120850615,0.00010353051,0.000033619195,0.00019084789,0.000045240864,0.000041242274,0.0014689016],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99855286,0.0006646611,0.00010968723,0.00024066324,0.0003387228,0.000093443196],"domain_scores_gemma":[0.994802,0.0033850172,0.0005665705,0.00027484234,0.0007254699,0.0002461327],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027273265,0.00081976136,0.0010160495,0.0005644181,0.0005983619,0.0009271747,0.0012220988,0.00090641325,0.0021753677],"category_scores_gemma":[0.010828505,0.00026437113,0.00037975542,0.00045140315,0.0010477349,0.0014326,0.0013738025,0.0015651962,0.00029936567],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011207366,0.000093125054,0.0006938706,0.00007738428,0.000032272812,0.00003422356,0.00006537685,0.9435049,0.0010551461,0.01489726,0.00037873196,0.03905565],"study_design_scores_gemma":[0.000014236923,0.00007602631,0.00010593861,0.000007143314,0.000005370159,0.000012867767,0.0000059649287,0.9940612,0.000421506,0.0049676066,0.0003152931,0.0000069739476],"about_ca_topic_score_codex":0.002625044,"about_ca_topic_score_gemma":0.0021231873,"teacher_disagreement_score":0.0027273265,"about_ca_system_score_codex":0.0015425046,"about_ca_system_score_gemma":0.00083100586,"threshold_uncertainty_score":0.014423668},"labels":[],"label_agreement":null},{"id":"W4399729340","doi":"10.1109/syscon61195.2024.10553598","title":"Deep Reinforcement Learning Agents for Decision Making for Gameplay","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Reinforcement learning; Computer science; Human–computer interaction; Artificial intelligence; Reinforcement; Psychology; Social psychology","score_opus":0.03155200176925009,"score_gpt":0.32377764727110686,"score_spread":0.29222564550185676,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399729340","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017470177,0.0006050787,0.97067386,0.0005434862,0.00008820848,0.00010238385,0.00008114058,0.0013327907,0.009102896],"genre_scores_gemma":[0.74463135,0.0004873571,0.24240243,0.0003342805,0.000050899675,0.0003372252,0.00016888426,0.00013954168,0.011448098],"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","domain_scores_codex":[0.99971575,0.00009301788,0.000017592427,0.00005112879,0.00007439891,0.00004803738],"domain_scores_gemma":[0.99925226,0.00043923547,0.00006519643,0.00005028217,0.00013505627,0.000057990226],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007462185,0.00075563707,0.0005751279,0.00030689806,0.0003794231,0.00095621555,0.0013955787,0.0010072658,0.005567947],"category_scores_gemma":[0.002519439,0.00041832784,0.00047877326,0.00023395351,0.0008102387,0.00096255913,0.00093101396,0.002133706,0.0009358534],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000094294264,0.00011949745,0.00068073993,0.00008819301,0.00004554575,0.00006925453,0.0000824087,0.88240147,0.0023803988,0.029224623,0.0024264858,0.082387134],"study_design_scores_gemma":[0.000008911082,0.000014980149,0.00004629064,0.000006529256,0.0000036030856,0.0000057912625,0.0000047000626,0.99137056,0.0003936906,0.0070921304,0.0010492699,0.0000035452294],"about_ca_topic_score_codex":0.008080744,"about_ca_topic_score_gemma":0.011524841,"teacher_disagreement_score":0.008080744,"about_ca_system_score_codex":0.0012946762,"about_ca_system_score_gemma":0.0012751428,"threshold_uncertainty_score":0.01862663},"labels":[],"label_agreement":null},{"id":"W4399828133","doi":"10.1007/978-3-031-54071-4_22","title":"Reinforcement Learning for Partially Observable Models","year":2024,"lang":"en","type":"book-chapter","venue":"Systems & control","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Observable; Reinforcement learning; Reinforcement; Computer science; Artificial intelligence; Psychology; Physics; Social psychology","score_opus":0.037490925067643545,"score_gpt":0.24043671886862947,"score_spread":0.20294579380098593,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399828133","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032326218,0.0034863488,0.9660266,0.0005108954,0.00016430799,0.000022274064,0.00011655748,0.0004729126,0.025967479],"genre_scores_gemma":[0.531077,0.010160993,0.35282502,0.00036323734,0.0005131467,0.0003596609,0.001016139,0.00035656974,0.10332829],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9997979,0.00006471359,0.0000080634545,0.00003176443,0.000082177765,0.00001541873],"domain_scores_gemma":[0.99969053,0.00022262784,0.000016708955,0.00003227789,0.000028885634,0.000009102766],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00030531923,0.00061310537,0.00073500245,0.00021198235,0.00019375315,0.00077969546,0.0007591365,0.0006222563,0.007030945],"category_scores_gemma":[0.0013097716,0.00035028646,0.00043563326,0.0003500147,0.00056999916,0.0011221103,0.000670508,0.0015084357,0.0011149718],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003292043,0.00004777208,0.00013691075,0.0002681352,0.00003912214,0.00008133189,0.00007760343,0.36460817,0.0016281168,0.43485534,0.0135884825,0.1846361],"study_design_scores_gemma":[0.0000094508505,0.000017080198,0.000080825965,0.000035741403,0.000006416428,0.000031259915,0.00000777213,0.7107845,0.0003943494,0.27767748,0.010945237,0.000009936555],"about_ca_topic_score_codex":0.0019982145,"about_ca_topic_score_gemma":0.002342145,"teacher_disagreement_score":0.007030945,"about_ca_system_score_codex":0.00070889783,"about_ca_system_score_gemma":0.0004492193,"threshold_uncertainty_score":0.023520827},"labels":[],"label_agreement":null},{"id":"W4399828205","doi":"10.1007/978-3-031-54071-4_21","title":"Reinforcement Learning for Fully Observable Models: Convergence to Near Optimality under Weak Feller Continuity","year":2024,"lang":"en","type":"book-chapter","venue":"Systems & control","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Observable; Reinforcement learning; Convergence (economics); Reinforcement; Computer science; Mathematics; Mathematical optimization; Artificial intelligence; Psychology; Economics; Physics; Social psychology","score_opus":0.04009775781671515,"score_gpt":0.2557781574063045,"score_spread":0.21568039958958934,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399828205","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024082776,0.0010513637,0.96111697,0.0010213385,0.00008999233,0.000051283027,0.00010635326,0.00026428382,0.0122156],"genre_scores_gemma":[0.82274383,0.001901533,0.15644093,0.00062067865,0.00022739048,0.0004068833,0.00036270978,0.00032649405,0.016969567],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9982039,0.0008961773,0.000091197704,0.00029223785,0.00036031057,0.00015619975],"domain_scores_gemma":[0.9851724,0.01283876,0.00057904655,0.00045357828,0.0005762326,0.00037988566],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045612245,0.0018287061,0.0030824535,0.001046138,0.00091203355,0.0024374942,0.0021501377,0.0027276676,0.005001202],"category_scores_gemma":[0.023959516,0.0012206115,0.001708366,0.0009061344,0.004403607,0.004883797,0.0042895623,0.005729779,0.0006440869],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013602723,0.000092605704,0.00035738538,0.0002771676,0.00008522796,0.00010538954,0.00027934104,0.33245713,0.0014085306,0.64119446,0.0022921902,0.021314451],"study_design_scores_gemma":[0.00001317835,0.00003461618,0.00004421279,0.000020594332,0.000007470462,0.000016097129,0.000012981624,0.6498581,0.00014582062,0.3495586,0.0002785044,0.000009810489],"about_ca_topic_score_codex":0.0046301205,"about_ca_topic_score_gemma":0.0024551607,"teacher_disagreement_score":0.005001202,"about_ca_system_score_codex":0.0025120154,"about_ca_system_score_gemma":0.0021871896,"threshold_uncertainty_score":0.024122357},"labels":[],"label_agreement":null},{"id":"W4400146922","doi":"10.1016/j.knosys.2024.112190","title":"DiffSkill: Improving Reinforcement Learning through diffusion-based skill denoiser for robotic manipulation","year":2024,"lang":"en","type":"article","venue":"Knowledge-Based Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"China Scholarship Council; Science and Technology Commission of Shanghai Municipality","keywords":"Reinforcement learning; Computer science; Reinforcement; Artificial intelligence; Materials science; Composite material","score_opus":0.025444150437695098,"score_gpt":0.27644692360467316,"score_spread":0.25100277316697805,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400146922","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02941159,0.00037002374,0.9651092,0.00020094372,0.00015203291,0.000089010624,0.000038765822,0.001834565,0.0027937852],"genre_scores_gemma":[0.79146206,0.00014401617,0.20088679,0.00026986634,0.00004557466,0.00015440193,0.00010062007,0.0002006787,0.0067359516],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995993,0.000069169124,0.00002202539,0.000100432226,0.00015185626,0.00005731331],"domain_scores_gemma":[0.9988638,0.0005812473,0.0000956019,0.00012740862,0.00023695164,0.00009499209],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012149729,0.00084831676,0.001087272,0.00053251674,0.0004329991,0.00055864366,0.0018956683,0.0013370826,0.0038456004],"category_scores_gemma":[0.0035023484,0.00040086303,0.00042307354,0.0002965785,0.0010031955,0.001078583,0.001991411,0.001969578,0.0006040239],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041580148,0.0004156288,0.0007319421,0.00018831276,0.00007211825,0.000105211715,0.000100916965,0.66631615,0.018436646,0.009675002,0.0031838084,0.30035847],"study_design_scores_gemma":[0.000027155304,0.00006307574,0.00006106787,0.0000048967845,0.0000055155137,0.000012268755,0.000003831325,0.99556655,0.0020768498,0.0018114692,0.00036159676,0.000005783981],"about_ca_topic_score_codex":0.0056145927,"about_ca_topic_score_gemma":0.0073963664,"teacher_disagreement_score":0.0056145927,"about_ca_system_score_codex":0.00069987826,"about_ca_system_score_gemma":0.0010647399,"threshold_uncertainty_score":0.012864828},"labels":[],"label_agreement":null},{"id":"W4400374283","doi":"10.48550/arxiv.2407.02231","title":"Safety-Driven Deep Reinforcement Learning Framework for Cobots: A Sim2Real Approach","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Canadian Institute of Steel Construction","keywords":"Reinforcement learning; Artificial intelligence; Reinforcement; Computer science; Psychology; Social psychology","score_opus":0.06038632336708043,"score_gpt":0.21454467826030085,"score_spread":0.15415835489322044,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400374283","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0147456275,0.00018596025,0.98005587,0.00016663392,0.00005289697,0.000047013895,0.000037080914,0.0014323683,0.0032765823],"genre_scores_gemma":[0.8106812,0.0001230127,0.18297689,0.00020247909,0.00004280039,0.00017986054,0.00014061572,0.00019915204,0.005454058],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99971145,0.00008042962,0.0000100714615,0.000063410516,0.00008113972,0.00005344286],"domain_scores_gemma":[0.99960953,0.00013453561,0.00004457859,0.00005224316,0.00010377138,0.000055413177],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008986343,0.0006971778,0.0006355667,0.0003280656,0.00029325482,0.000617579,0.0018110556,0.0009601352,0.0033954391],"category_scores_gemma":[0.0011889391,0.00035574168,0.00050211675,0.00015632721,0.00069406035,0.00061270554,0.0013371162,0.0011909219,0.00060407835],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007018585,0.000064401735,0.0004605977,0.0000454366,0.000026613265,0.00006117278,0.000031729815,0.95237094,0.0017773947,0.0055076876,0.0008229413,0.038760833],"study_design_scores_gemma":[0.0000033266083,0.000016224805,0.000022454447,0.0000017124761,0.0000014665644,0.0000048499637,0.0000019626898,0.9985921,0.00019882698,0.000895696,0.00026010664,0.0000012997651],"about_ca_topic_score_codex":0.005913796,"about_ca_topic_score_gemma":0.0062286872,"teacher_disagreement_score":0.005913796,"about_ca_system_score_codex":0.0008753893,"about_ca_system_score_gemma":0.0013114985,"threshold_uncertainty_score":0.011758745},"labels":[],"label_agreement":null},{"id":"W4400947559","doi":"10.3390/app14156432","title":"Environmental-Impact-Based Multi-Agent Reinforcement Learning","year":2024,"lang":"en","type":"article","venue":"Applied Sciences","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Brock University","funders":"Alliance de recherche numérique du Canada","keywords":"Computer science","score_opus":0.027554366087087664,"score_gpt":0.2775524108106826,"score_spread":0.24999804472359494,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400947559","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12972718,0.00020491808,0.86588246,0.00028702494,0.000058094567,0.00012907146,0.00002636167,0.00050543353,0.0031794088],"genre_scores_gemma":[0.9612188,0.00004455377,0.03778508,0.00008634758,0.000014931971,0.00008853218,0.000022948952,0.000012482247,0.0007264164],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994147,0.00023086261,0.000029914885,0.00011689868,0.00013816077,0.000069464906],"domain_scores_gemma":[0.99843735,0.00074527494,0.0002700134,0.00013093854,0.00028223154,0.00013420994],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014915577,0.0008386816,0.00084374106,0.0003834813,0.00028698292,0.00045285383,0.0015805869,0.00064964435,0.0010799433],"category_scores_gemma":[0.0037377914,0.00023285637,0.00036027306,0.00019644246,0.00077007664,0.0006381046,0.0010051276,0.00087955507,0.00016553515],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007834528,0.0002146721,0.0030400977,0.000054981767,0.00007540717,0.00007933594,0.00008144927,0.93867964,0.0030144711,0.0038123133,0.00038293717,0.050486334],"study_design_scores_gemma":[0.000012859932,0.000059018308,0.00020503553,0.000003259765,0.000006630239,0.000012069986,0.0000042780816,0.9977847,0.0004607913,0.0013053108,0.00014109502,0.0000050018466],"about_ca_topic_score_codex":0.0019783634,"about_ca_topic_score_gemma":0.0021355688,"teacher_disagreement_score":0.0019783634,"about_ca_system_score_codex":0.00064788986,"about_ca_system_score_gemma":0.00071172835,"threshold_uncertainty_score":0.007888198},"labels":[],"label_agreement":null},{"id":"W4401023823","doi":"10.24963/ijcai.2024/587","title":"Towards Debiased Generalized Category Discovery","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Natural Science Foundation of Jiangsu Province; National Natural Science Foundation of China; Government of Jiangsu Province; Hong Kong Baptist University","keywords":"Computer science; Set (abstract data type); Artificial intelligence; Programming language","score_opus":0.022507952834753153,"score_gpt":0.26828201683960556,"score_spread":0.2457740640048524,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401023823","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06596802,0.001040743,0.92889845,0.0006083312,0.00007135952,0.00008560171,0.00013509282,0.0017364004,0.0014559751],"genre_scores_gemma":[0.6238965,0.00036293553,0.36954084,0.0010874862,0.00014977805,0.00015025947,0.0009177043,0.0002591232,0.0036353744],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9964072,0.0010163429,0.00016662951,0.0011822068,0.0009208042,0.0003068345],"domain_scores_gemma":[0.99008733,0.0040725796,0.0009348834,0.0024415723,0.0019971614,0.00046644086],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0057933438,0.0014167036,0.002545886,0.00225448,0.0012214966,0.0018265522,0.0037121943,0.0023982008,0.0014357106],"category_scores_gemma":[0.014580932,0.0006296953,0.0011027919,0.0019416728,0.0022388988,0.0037882729,0.004683207,0.0031261782,0.0007431917],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000654033,0.00024109051,0.009456068,0.00023616149,0.0001878505,0.0002126661,0.00050610775,0.1275485,0.0092916945,0.01912246,0.008208722,0.8243347],"study_design_scores_gemma":[0.000046338213,0.00016195774,0.00090129185,0.00003443117,0.000039076695,0.00021632963,0.0001252458,0.9478309,0.005788445,0.04165747,0.0031628243,0.00003558924],"about_ca_topic_score_codex":0.0041939532,"about_ca_topic_score_gemma":0.0043356293,"teacher_disagreement_score":0.0057933438,"about_ca_system_score_codex":0.001434329,"about_ca_system_score_gemma":0.0025748108,"threshold_uncertainty_score":0.030638516},"labels":[],"label_agreement":null},{"id":"W4401023970","doi":"10.24963/ijcai.2024/427","title":"Α Descent-based Method on the Duality Gap for Solving Zero-sum Games","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; Korea Advanced Institute of Science and Technology","keywords":"Reinforcement learning; Diversification (marketing strategy); Computer science; Reinforcement; Artificial intelligence; Business; Engineering; Marketing","score_opus":0.0741761737187263,"score_gpt":0.3404508027115267,"score_spread":0.2662746289928004,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401023970","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007976796,0.00013506826,0.9884277,0.000115693634,0.00003902093,0.00006708975,0.000017078619,0.00025469658,0.002966861],"genre_scores_gemma":[0.3437363,0.00026806764,0.64916885,0.00031940153,0.000066896544,0.0004412251,0.00011022173,0.00027604276,0.005612951],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9993795,0.00025079557,0.000027444874,0.00008926006,0.00016717757,0.00008584515],"domain_scores_gemma":[0.9989033,0.00065171864,0.00008220107,0.00008349205,0.00019281685,0.00008652563],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017699593,0.0010430212,0.0013074043,0.00070961495,0.00051460013,0.0009639848,0.0012328988,0.0011275074,0.0030900605],"category_scores_gemma":[0.0046033626,0.00047798804,0.0006709756,0.00048962835,0.0011376109,0.0009277004,0.0014560845,0.0017745035,0.0008564616],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010291284,0.00010569944,0.00051194156,0.00012889622,0.000035057736,0.00007550836,0.000089176,0.8755225,0.0023501974,0.05862094,0.0021609261,0.060296115],"study_design_scores_gemma":[0.000015592928,0.000035232457,0.00003318155,0.000010038099,0.000003401532,0.000013528185,0.0000068792956,0.98933905,0.00047946948,0.009416025,0.00064381,0.000003734876],"about_ca_topic_score_codex":0.0024946327,"about_ca_topic_score_gemma":0.0021081634,"teacher_disagreement_score":0.0030900605,"about_ca_system_score_codex":0.00105413,"about_ca_system_score_gemma":0.0021244627,"threshold_uncertainty_score":0.010337234},"labels":[],"label_agreement":null},{"id":"W4401024442","doi":"10.24963/ijcai.2024/507","title":"Reasoning About Causal Knowledge in Nondeterministic Domains","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; University of Regina","funders":"Youth Innovation Promotion Association; Tencent; National Key Research and Development Program of China; Youth Innovation Promotion Association of the Chinese Academy of Sciences; Chinese Academy of Sciences; National Natural Science Foundation of China","keywords":"Reinforcement learning; Computer science; Operator (biology); Artificial intelligence","score_opus":0.015185618027503484,"score_gpt":0.2941535458412524,"score_spread":0.2789679278137489,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401024442","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019159654,0.00036505185,0.97551584,0.0013827741,0.00006217892,0.0000694435,0.00013201256,0.00042668343,0.0028863978],"genre_scores_gemma":[0.6782189,0.00086552266,0.3177368,0.00043054434,0.00020955858,0.00016709442,0.00045080294,0.00011757222,0.00180326],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9922064,0.0029836842,0.000765673,0.0013669814,0.0020060728,0.00067124475],"domain_scores_gemma":[0.96527815,0.027337423,0.0026441254,0.0023804426,0.0015978616,0.00076197006],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010207614,0.0011286282,0.0012410389,0.0025850085,0.0022580374,0.0046256874,0.002703861,0.002064132,0.002386267],"category_scores_gemma":[0.034122232,0.0011605959,0.002880474,0.0015750058,0.006349773,0.011553705,0.0049214014,0.0052330797,0.00025344352],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000117348165,0.00008607265,0.0021967927,0.00024397798,0.00015119567,0.0007967384,0.0008139963,0.26411638,0.0014028351,0.7006339,0.0010433241,0.028397422],"study_design_scores_gemma":[0.000025745547,0.00001846728,0.00024406353,0.000038910886,0.00005112818,0.00008625277,0.000103540835,0.2940516,0.0014691316,0.7012516,0.0026250302,0.00003454781],"about_ca_topic_score_codex":0.01118766,"about_ca_topic_score_gemma":0.011418813,"teacher_disagreement_score":0.01118766,"about_ca_system_score_codex":0.0027507178,"about_ca_system_score_gemma":0.0034394066,"threshold_uncertainty_score":0.05398363},"labels":[],"label_agreement":null},{"id":"W4401024943","doi":"10.24963/ijcai.2024/964","title":"DGL: Dynamic Global-Local Information Aggregation for Scalable VRP Generalization with Self-Improvement Learning","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"BC Research (Canada)","funders":"Grantová Agentura České Republiky; Ministerstvo Školství, Mládeže a Tělovýchovy","keywords":"Computer science; Imperfect; Perfect information; Algorithm; Mathematics; Mathematical economics","score_opus":0.004381780029235462,"score_gpt":0.22670872808770698,"score_spread":0.22232694805847153,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401024943","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021512872,0.00069583714,0.9718239,0.0005586156,0.00007861487,0.00012783447,0.00016004674,0.002660821,0.0023815634],"genre_scores_gemma":[0.7025317,0.0003541704,0.2914015,0.00095278997,0.00012337304,0.0004556444,0.0007817215,0.00044766298,0.0029514146],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99903893,0.00029024165,0.0000588519,0.00026614673,0.00022929831,0.00011665596],"domain_scores_gemma":[0.9974233,0.001424057,0.00026317398,0.00038548835,0.00034586934,0.00015816624],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027280736,0.0017269186,0.0018749331,0.000986494,0.0005988705,0.0010840301,0.0033581164,0.0016766256,0.002355249],"category_scores_gemma":[0.006464341,0.0008196848,0.0010131723,0.0009392044,0.0014063797,0.00211549,0.0034979393,0.0028263195,0.0006779262],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000048228685,0.00010901118,0.0009462182,0.000083745654,0.000051074196,0.000062767474,0.00006358566,0.9284842,0.00079810666,0.0038509378,0.0029741721,0.06252798],"study_design_scores_gemma":[0.000009019588,0.00002069803,0.000034027336,0.0000045177,0.0000034292825,0.0000055055616,0.0000038530843,0.99775285,0.0001493186,0.00179556,0.00021880574,0.0000025169354],"about_ca_topic_score_codex":0.0072368123,"about_ca_topic_score_gemma":0.0084017515,"teacher_disagreement_score":0.0072368123,"about_ca_system_score_codex":0.0016738719,"about_ca_system_score_gemma":0.0017486637,"threshold_uncertainty_score":0.014427602},"labels":[],"label_agreement":null},{"id":"W4401157207","doi":"10.21203/rs.3.rs-4678044/v1","title":"Speeding up hierarchical reinforcement learning using state-independent temporal skills","year":2024,"lang":"en","type":"preprint","venue":"Research Square","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Reinforcement learning; Reinforcement; Computer science; State (computer science); Artificial intelligence; Psychology; Social psychology; Algorithm","score_opus":0.07600609675111851,"score_gpt":0.39778746747497096,"score_spread":0.32178137072385243,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401157207","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13962531,0.00018431585,0.85353,0.00023948087,0.00009259555,0.000110367364,0.00006426464,0.0016117571,0.0045419554],"genre_scores_gemma":[0.9281188,0.000044685108,0.0697998,0.00006729334,0.000017173577,0.000078960344,0.000053085936,0.00005239327,0.0017677865],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996518,0.000063454696,0.000019977879,0.00009787864,0.000095624324,0.00007124599],"domain_scores_gemma":[0.9983012,0.0010623469,0.00012184776,0.00021168325,0.00016419144,0.00013868764],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008454053,0.00062523165,0.0007101876,0.00030360548,0.0003104317,0.0005243474,0.0010984391,0.00077865564,0.0044351923],"category_scores_gemma":[0.004282263,0.0004097004,0.00034109174,0.00025074676,0.000658693,0.0012502153,0.0013554585,0.001618193,0.0005461622],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00054620445,0.0005452037,0.0017723466,0.00017771246,0.00007654252,0.000115113195,0.00013888322,0.68300873,0.020697182,0.01609353,0.0026030873,0.27422547],"study_design_scores_gemma":[0.000026775313,0.000041474923,0.00010895085,0.0000030630392,0.0000053003414,0.0000068357276,0.0000038661847,0.9950448,0.0011947986,0.0034358618,0.00012549284,0.0000027055778],"about_ca_topic_score_codex":0.004968215,"about_ca_topic_score_gemma":0.0066436017,"teacher_disagreement_score":0.004968215,"about_ca_system_score_codex":0.000596894,"about_ca_system_score_gemma":0.0012284922,"threshold_uncertainty_score":0.014837146},"labels":[],"label_agreement":null},{"id":"W4401413897","doi":"10.1109/icra57147.2024.10610017","title":"Towards Real-World Efficiency: Domain Randomization in Reinforcement Learning for Pre-Capture of Free-Floating Moving Targets by Autonomous Robots","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Reinforcement learning; Robot; Computer science; Domain (mathematical analysis); Randomization; Artificial intelligence; Mathematics","score_opus":0.009420998089325656,"score_gpt":0.26129184082772616,"score_spread":0.2518708427384005,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401413897","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030075224,0.00013639756,0.96763,0.00015432214,0.000021017404,0.0000687661,0.000013134493,0.0003907981,0.0015104251],"genre_scores_gemma":[0.8732483,0.000094342504,0.12481087,0.00015392996,0.00002020894,0.00017871229,0.000034564502,0.00010753333,0.0013515687],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992005,0.00035223813,0.00003799538,0.00015846967,0.00016160877,0.00008920799],"domain_scores_gemma":[0.99654514,0.0021476957,0.0004264003,0.0004195963,0.00030127048,0.00015989144],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002966966,0.00092957844,0.0007737258,0.00033015778,0.00032879328,0.00076481217,0.0012252955,0.0008505621,0.0014449258],"category_scores_gemma":[0.008751682,0.00040235498,0.0003575389,0.00020346286,0.0015906018,0.0014095616,0.0015328914,0.0017191899,0.0002912462],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014544974,0.0001186188,0.0010054401,0.00007494661,0.00003950095,0.00005374758,0.00008101653,0.9414841,0.005753766,0.012306898,0.0004524597,0.038484033],"study_design_scores_gemma":[0.000016213526,0.000056878067,0.00010603344,0.0000075950984,0.000003890486,0.000010410864,0.000006443848,0.9933838,0.0011684811,0.004993556,0.00024171277,0.0000049235537],"about_ca_topic_score_codex":0.0020837847,"about_ca_topic_score_gemma":0.0018050118,"teacher_disagreement_score":0.002966966,"about_ca_system_score_codex":0.001059148,"about_ca_system_score_gemma":0.0014093482,"threshold_uncertainty_score":0.015691042},"labels":[],"label_agreement":null},{"id":"W4401422722","doi":"10.1145/3643658.3643919","title":"A Behavior-driven Development and Reinforcement Learning approach for videogame automated testing","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Reinforcement learning; Computer science; Reinforcement; Development (topology); Artificial intelligence; Human–computer interaction; Engineering","score_opus":0.04646166824848585,"score_gpt":0.2825911991807369,"score_spread":0.23612953093225103,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401422722","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0070281555,0.000043486485,0.98984575,0.00013602215,0.000011246169,0.00022717139,0.00003917445,0.0014117773,0.0012571723],"genre_scores_gemma":[0.27328977,0.00007619053,0.7240331,0.00015552327,0.0000127925005,0.000629975,0.0001393801,0.00026373856,0.001399508],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99639195,0.0016069068,0.00020450032,0.00054929586,0.0010124466,0.00023477044],"domain_scores_gemma":[0.9931564,0.0041024876,0.00064728514,0.00076957524,0.0010730025,0.00025127415],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003622403,0.0011769123,0.000610129,0.0011059712,0.00039191774,0.0011318517,0.0025910363,0.00092346605,0.0019404644],"category_scores_gemma":[0.011573833,0.00071299245,0.0010973059,0.00036909457,0.0018380563,0.0011149453,0.001743358,0.0019056078,0.00038881245],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016478887,0.0006763944,0.005293683,0.00039344846,0.00012257368,0.00037471147,0.0004915255,0.72499895,0.016493903,0.048598677,0.0017671706,0.20062423],"study_design_scores_gemma":[0.00002592553,0.000114842864,0.00030406035,0.000028237011,0.000015935648,0.000067352055,0.000021131427,0.98227143,0.00502145,0.010188939,0.0019234128,0.000017179418],"about_ca_topic_score_codex":0.0054394635,"about_ca_topic_score_gemma":0.006089124,"teacher_disagreement_score":0.0054394635,"about_ca_system_score_codex":0.0017209846,"about_ca_system_score_gemma":0.0030323046,"threshold_uncertainty_score":0.01915729},"labels":[],"label_agreement":null},{"id":"W4401540274","doi":"10.1109/infocomwkshps61880.2024.10620892","title":"Cascade: Enhancing Reinforcement Learning with Curriculum Federated Learning and Interference Avoidance — A Case Study in Adaptive Bitrate Selection","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Cascade; Selection (genetic algorithm); Interference (communication); Artificial intelligence; Machine learning; Computer network; Engineering","score_opus":0.014591500648883905,"score_gpt":0.2666688859663599,"score_spread":0.252077385317476,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401540274","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15952733,0.00059625204,0.83271337,0.00041877682,0.000085231404,0.00022480192,0.000056470497,0.0017939094,0.004583888],"genre_scores_gemma":[0.9172566,0.00010936907,0.0810889,0.00014043374,0.000022968376,0.000090508416,0.000035909143,0.00004592521,0.0012093188],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99910104,0.00031209888,0.000039846076,0.00019227598,0.00022125787,0.00013345854],"domain_scores_gemma":[0.99713093,0.00176992,0.00019477496,0.00028236912,0.00040585682,0.00021612404],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026913546,0.00093604106,0.0008502043,0.00048377019,0.00045116094,0.00070429617,0.0017259924,0.001092434,0.0014880685],"category_scores_gemma":[0.0076047718,0.0002598753,0.0004181805,0.00040619113,0.0010180246,0.0012529059,0.001232884,0.0015038606,0.00022946097],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038195314,0.00040667618,0.002927235,0.000102295584,0.000052758132,0.00026843767,0.00012830472,0.86857814,0.0048642843,0.0057867025,0.001250989,0.1152522],"study_design_scores_gemma":[0.000028659711,0.00017491526,0.00021312953,0.0000061290893,0.000008761934,0.000047948495,0.000011851752,0.99512154,0.0019120702,0.00205818,0.00040923696,0.0000074851146],"about_ca_topic_score_codex":0.0049397685,"about_ca_topic_score_gemma":0.0037429444,"teacher_disagreement_score":0.0049397685,"about_ca_system_score_codex":0.00086366176,"about_ca_system_score_gemma":0.0011918015,"threshold_uncertainty_score":0.01423341},"labels":[],"label_agreement":null},{"id":"W4401541260","doi":"10.1109/infocomwkshps61880.2024.10620663","title":"Maximizing the Social Welfare of Decentralized Knowledge Inference Through Evolutionary Game","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Inference; Computer science; Game theory; Welfare; Artificial intelligence; Social Welfare; Evolutionary game theory; Mathematical economics; Economics; Political science","score_opus":0.03654919925995269,"score_gpt":0.3297239862637389,"score_spread":0.2931747870037862,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401541260","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2601177,0.00019876387,0.72988784,0.00199277,0.00003853283,0.00015211006,0.000103837534,0.00017291549,0.0073355413],"genre_scores_gemma":[0.9662859,0.000099753044,0.031129576,0.00012398696,0.000017909022,0.00010893135,0.000045792956,0.000026010004,0.0021622432],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99838376,0.0009452136,0.000045845667,0.00024862835,0.00018056181,0.00019604864],"domain_scores_gemma":[0.9943129,0.0041706995,0.0005230092,0.00032371812,0.0002603965,0.000409302],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029582304,0.000863419,0.0013428702,0.00059316965,0.0007557307,0.0014641902,0.0016500722,0.0015864854,0.0015864099],"category_scores_gemma":[0.0115718525,0.00048387877,0.0005246671,0.0005068236,0.001745026,0.0021593475,0.0016356893,0.0012100782,0.00017012446],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013485317,0.0001227041,0.0011416427,0.000063352025,0.000058105263,0.00016986472,0.00015497662,0.93924254,0.0014560949,0.044201184,0.00091415376,0.012340476],"study_design_scores_gemma":[0.00001835149,0.000020126547,0.00010674811,0.0000040027776,0.0000070011024,0.000016392773,0.000020815556,0.9784686,0.00015964232,0.021008082,0.0001655744,0.000004607994],"about_ca_topic_score_codex":0.0031528478,"about_ca_topic_score_gemma":0.0023208868,"teacher_disagreement_score":0.0031528478,"about_ca_system_score_codex":0.0019971565,"about_ca_system_score_gemma":0.0016874815,"threshold_uncertainty_score":0.015644789},"labels":[],"label_agreement":null},{"id":"W4401567462","doi":"10.1109/jiot.2024.3443701","title":"Energy-Efficient Joint Optimization of Sensing and Computation in MEC-Assisted IoT Using Mean-Field Game","year":2024,"lang":"en","type":"article","venue":"IEEE Internet of Things Journal","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"Toyota Motor Corporation; National Natural Science Foundation of China","keywords":"Computer science; Server; Energy consumption; Computation offloading; Distributed computing; Computation; Efficient energy use; Computational complexity theory; Optimization problem; Edge computing; Mobile edge computing; Markov decision process; Field (mathematics); Internet of Things; Enhanced Data Rates for GSM Evolution; Markov process; Computer network; Artificial intelligence; Algorithm; Embedded system","score_opus":0.031434826967558574,"score_gpt":0.2754179182907363,"score_spread":0.24398309132317775,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401567462","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08071963,0.0004112697,0.9105447,0.00074905454,0.00009198318,0.00008924273,0.0000975826,0.00018549651,0.007110988],"genre_scores_gemma":[0.97873014,0.00013383829,0.018726379,0.00011310087,0.00001699179,0.0000857549,0.000039225306,0.000020000783,0.0021345057],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994708,0.00019171221,0.000018803114,0.00011110213,0.000087937384,0.000119702905],"domain_scores_gemma":[0.99867487,0.00094311306,0.00011562378,0.000038071958,0.00012536488,0.00010301834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011107138,0.0010010645,0.001398308,0.00035843323,0.00042586387,0.0009553618,0.0010916366,0.0012032613,0.0015058697],"category_scores_gemma":[0.0025304083,0.00050929777,0.0006037031,0.00039200718,0.00114456,0.0009932384,0.0010375581,0.0010736893,0.00014323583],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000052788197,0.00002416391,0.00034343105,0.0000278717,0.000019855926,0.00007306089,0.000021043239,0.9883301,0.0006112639,0.0069699995,0.0004029662,0.003123454],"study_design_scores_gemma":[0.000005655099,0.000009998645,0.00004050815,0.0000016317063,0.000002933063,0.0000060126845,0.000003836103,0.9979942,0.000057777583,0.0018193464,0.000056014378,0.0000021608741],"about_ca_topic_score_codex":0.006613898,"about_ca_topic_score_gemma":0.004852817,"teacher_disagreement_score":0.006613898,"about_ca_system_score_codex":0.0010712775,"about_ca_system_score_gemma":0.0013589057,"threshold_uncertainty_score":0.013150811},"labels":[],"label_agreement":null},{"id":"W4401687215","doi":"10.1109/access.2024.3446310","title":"Finding the Optimal Security Policies for Autonomous Cyber Operations With Competitive Reinforcement Learning","year":2024,"lang":"en","type":"article","venue":"IEEE Access","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Defence Research and Development Canada; Royal Military College of Canada","funders":"","keywords":"Reinforcement learning; Computer science; Computer security; Artificial intelligence","score_opus":0.030520465487766258,"score_gpt":0.3178122432086103,"score_spread":0.287291777720844,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401687215","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22948243,0.00029112925,0.76112026,0.00058849284,0.0000658206,0.00017270664,0.00005256777,0.00035043177,0.007876159],"genre_scores_gemma":[0.9589724,0.000054036256,0.039786287,0.000102885446,0.00001076368,0.00007842451,0.000031691896,0.000021917696,0.00094165263],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99951565,0.00019319852,0.000025923146,0.00007750115,0.00010689715,0.000080768805],"domain_scores_gemma":[0.9963527,0.0026555278,0.00031037247,0.00013945122,0.00037902358,0.00016293302],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016596634,0.000858703,0.0007079938,0.0005790107,0.0004398352,0.0007227686,0.0009793638,0.001038684,0.00161683],"category_scores_gemma":[0.007493986,0.00042951526,0.00044135872,0.00025757359,0.0012651851,0.0007355851,0.00078177423,0.001332456,0.0001951033],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000028827048,0.000037974376,0.0006096988,0.000019575762,0.000013404147,0.000021882975,0.000033758784,0.9902872,0.00037096048,0.003523548,0.00014358194,0.0049095843],"study_design_scores_gemma":[0.000003531925,0.000013530797,0.00003143839,0.0000016242632,0.0000014317773,0.0000025252627,0.0000028230054,0.9988997,0.00008633204,0.0009188074,0.000036901063,0.0000013294958],"about_ca_topic_score_codex":0.010915447,"about_ca_topic_score_gemma":0.009777697,"teacher_disagreement_score":0.010915447,"about_ca_system_score_codex":0.0013613766,"about_ca_system_score_gemma":0.0016277226,"threshold_uncertainty_score":0.02170384},"labels":[],"label_agreement":null},{"id":"W4401883986","doi":"10.4018/979-8-3693-7668-3.ch015","title":"Harnessing Predictive Analytics for Workforce Optimization in a Transhuman Age","year":2024,"lang":"en","type":"book-chapter","venue":"Advances in human resources management and organizational development book series","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Yorkville University","funders":"","keywords":"Workforce; Analytics; Predictive analytics; Computer science; Data science; Political science","score_opus":0.012570200725336402,"score_gpt":0.23227836975743946,"score_spread":0.21970816903210305,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401883986","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02050427,0.0078036683,0.9160572,0.005174874,0.0005200771,0.00008872762,0.00052289054,0.00285459,0.046473704],"genre_scores_gemma":[0.4988918,0.016643457,0.4493306,0.0015589431,0.0007301527,0.00024157298,0.0017516821,0.00061815564,0.030233612],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9996345,0.00009777907,0.000013158724,0.000080523285,0.00014607438,0.00002795762],"domain_scores_gemma":[0.99899584,0.00067249243,0.0000624403,0.00009620899,0.00012559188,0.00004736487],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00086384197,0.00083549105,0.0005903907,0.0009081668,0.0003726885,0.0024889081,0.0009786297,0.0007738984,0.0042244145],"category_scores_gemma":[0.0023283227,0.00029114736,0.000556183,0.0011926964,0.0005895045,0.002250288,0.0015496999,0.0015737937,0.0017398662],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000078414414,0.00015837187,0.0047583594,0.00035224526,0.00011225611,0.00029027107,0.0006010095,0.23031598,0.004634132,0.074167,0.025315484,0.6592165],"study_design_scores_gemma":[0.000013138746,0.00011310305,0.0028492496,0.00027635734,0.000047109846,0.00020860069,0.00046558204,0.6984678,0.0028298262,0.21691492,0.077747084,0.00006718597],"about_ca_topic_score_codex":0.003177245,"about_ca_topic_score_gemma":0.0038517155,"teacher_disagreement_score":0.0042244145,"about_ca_system_score_codex":0.0007158455,"about_ca_system_score_gemma":0.00081968715,"threshold_uncertainty_score":0.014132082},"labels":[],"label_agreement":null},{"id":"W4401911421","doi":"10.1080/09515089.2024.2393681","title":"Debt-free intelligence: ecological information in minds and machines","year":2024,"lang":"en","type":"article","venue":"Philosophical Psychology","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University; University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Psychology; Cognitive science; Ecology; Biology","score_opus":0.03319261259617537,"score_gpt":0.3323698577306866,"score_spread":0.29917724513451127,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401911421","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.063474536,0.010791924,0.73647827,0.042407036,0.0007014089,0.00009349897,0.00045551185,0.00061882765,0.14497907],"genre_scores_gemma":[0.8986487,0.0054429895,0.079233,0.0026652382,0.00096480263,0.00019013596,0.0002877236,0.00018641929,0.012381048],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.999114,0.0003008922,0.00005195271,0.00019724676,0.0002551424,0.00008083434],"domain_scores_gemma":[0.9960556,0.0019866894,0.00038995768,0.0008195606,0.00042648934,0.00032169072],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022844991,0.00043323793,0.00066027755,0.001540683,0.0015835372,0.0043418068,0.0015201053,0.0020912068,0.004535589],"category_scores_gemma":[0.011341293,0.000446366,0.0006571937,0.0015735818,0.009872106,0.016247945,0.0039574816,0.0028213274,0.00051046396],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000009757945,0.000007426449,0.0004056614,0.000040753723,0.000012365599,0.00004886111,0.0002923882,0.0033217524,0.0001573749,0.98285127,0.0013711086,0.011481394],"study_design_scores_gemma":[0.000003135492,0.0000027559322,0.00017792094,0.000011673788,0.0000030933757,0.000028599297,0.000040372113,0.004786906,0.000047926475,0.9910409,0.0038512982,0.0000055095607],"about_ca_topic_score_codex":0.0019468322,"about_ca_topic_score_gemma":0.0016652065,"teacher_disagreement_score":0.004535589,"about_ca_system_score_codex":0.0015681463,"about_ca_system_score_gemma":0.0011335793,"threshold_uncertainty_score":0.015173018},"labels":[],"label_agreement":null},{"id":"W4401971650","doi":"10.1016/j.nlm.2024.107974","title":"A bio-inspired reinforcement learning model that accounts for fast adaptation after punishment","year":2024,"lang":"en","type":"article","venue":"Neurobiology of Learning and Memory","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge; Mount Royal University","funders":"","keywords":"Reinforcement learning; Punishment (psychology); Adaptation (eye); Reinforcement; Artificial intelligence; Learning rule; Action (physics); Computer science; Psychology; Machine learning; Cognitive psychology; Artificial neural network; Cognitive science; Neuroscience; Social psychology","score_opus":0.02508569619655849,"score_gpt":0.25931858521347206,"score_spread":0.23423288901691358,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401971650","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038580433,0.00050510303,0.9398286,0.0014003009,0.00023300183,0.000060425795,0.00019623249,0.00031604993,0.018879944],"genre_scores_gemma":[0.890851,0.00060624897,0.08255624,0.0004268811,0.00012898089,0.00026361557,0.00013253778,0.00009606396,0.02493838],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998846,0.00002628065,0.0000059986596,0.000029599742,0.000033781307,0.000019845167],"domain_scores_gemma":[0.9997423,0.000094229385,0.000046357378,0.000024349634,0.000057876587,0.00003483692],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00039194556,0.00049507065,0.0006453979,0.00031465807,0.0003410095,0.00065563037,0.0013202953,0.0013367739,0.0027504354],"category_scores_gemma":[0.001025975,0.00022754312,0.00060762436,0.00023097201,0.0007606003,0.00089292653,0.00045071047,0.0012321714,0.0005686098],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000030423435,0.000041983938,0.0006387059,0.000039157152,0.000032376,0.00012923665,0.00004796889,0.90313,0.0026968303,0.07727801,0.0016138699,0.014321439],"study_design_scores_gemma":[0.00000857916,0.000013217241,0.000095279465,0.000002786627,0.0000055066666,0.000028593735,0.000001656681,0.98350304,0.00013732415,0.015492915,0.0007059007,0.000005219736],"about_ca_topic_score_codex":0.0035598068,"about_ca_topic_score_gemma":0.0028605966,"teacher_disagreement_score":0.0035598068,"about_ca_system_score_codex":0.00071912544,"about_ca_system_score_gemma":0.000721248,"threshold_uncertainty_score":0.009201109},"labels":[],"label_agreement":null},{"id":"W4402210774","doi":"10.1007/978-3-031-70903-6_16","title":"How to Better Fit Reinforcement Learning for Pentesting: A New Hierarchical Approach","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Reinforcement learning; Artificial intelligence; Machine learning","score_opus":0.03748018830220856,"score_gpt":0.26538449634167427,"score_spread":0.2279043080394657,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402210774","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004678296,0.00023407848,0.9903931,0.00032358422,0.00008000049,0.00004026554,0.00004070655,0.00090889324,0.0033010517],"genre_scores_gemma":[0.1875816,0.0002729496,0.8032508,0.00039493857,0.0000835305,0.00014556534,0.00015122491,0.00066478236,0.0074546197],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991721,0.00025840016,0.000053169562,0.00020275312,0.00019343258,0.00012027133],"domain_scores_gemma":[0.9985228,0.0006218316,0.00008942576,0.0004160154,0.0002414848,0.0001083853],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011632611,0.0008940043,0.0012985275,0.00046913183,0.00054970704,0.0011652707,0.0022693726,0.0014610487,0.011104683],"category_scores_gemma":[0.005796945,0.0005665017,0.0010460274,0.00057970756,0.00084973744,0.0033046706,0.0020907829,0.0029506946,0.0018156845],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016067251,0.00021787983,0.0008426856,0.00024366629,0.00011117473,0.00011033976,0.0001701271,0.519734,0.005465424,0.072175324,0.0093315225,0.39143717],"study_design_scores_gemma":[0.000016366474,0.000038103706,0.00007766568,0.000017091927,0.00001633211,0.000021550368,0.000017376587,0.9588696,0.00092970586,0.037638288,0.002347161,0.000010786591],"about_ca_topic_score_codex":0.0062777414,"about_ca_topic_score_gemma":0.009621477,"teacher_disagreement_score":0.011104683,"about_ca_system_score_codex":0.0009387751,"about_ca_system_score_gemma":0.0015742307,"threshold_uncertainty_score":0.037148833},"labels":[],"label_agreement":null},{"id":"W4402352014","doi":"10.1109/ijcnn60899.2024.10650768","title":"Conservative In-Distribution Q-Learning for Offline Reinforcement Learning","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"National Natural Science Foundation of China","keywords":"Reinforcement learning; Computer science; Distribution (mathematics); Reinforcement; Q-learning; Artificial intelligence; Machine learning; Engineering; Mathematics; Structural engineering","score_opus":0.02239324758151313,"score_gpt":0.2819598571834541,"score_spread":0.259566609601941,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402352014","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007855391,0.00017244412,0.98965365,0.00020091089,0.00003735802,0.000058550148,0.000036494013,0.0006944338,0.0012908585],"genre_scores_gemma":[0.7097112,0.0002329353,0.2849195,0.00064549124,0.00009566724,0.00048334783,0.00032057485,0.0003346637,0.003256631],"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","domain_scores_codex":[0.99763274,0.0010892852,0.00011687267,0.00040623226,0.00053594855,0.00021886893],"domain_scores_gemma":[0.9912272,0.0061455774,0.0005633441,0.0009965401,0.0007301907,0.0003372302],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005031079,0.0013124201,0.0015285049,0.0005943187,0.0005322823,0.0013286776,0.002500588,0.001349881,0.0035070928],"category_scores_gemma":[0.019214546,0.0005740287,0.00053649495,0.0005491687,0.001918705,0.0016564693,0.0022168707,0.0034965584,0.0007038252],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032055142,0.00023697368,0.0020452456,0.00022743575,0.00006825786,0.00010060074,0.00012243161,0.8476379,0.002728555,0.040000856,0.0032785481,0.10323273],"study_design_scores_gemma":[0.000020947371,0.000048967562,0.00008787103,0.0000110991,0.000004322545,0.000013449742,0.000005716456,0.98508745,0.0007507808,0.013393622,0.0005706924,0.000005095865],"about_ca_topic_score_codex":0.0024956046,"about_ca_topic_score_gemma":0.0024518417,"teacher_disagreement_score":0.005031079,"about_ca_system_score_codex":0.0016253028,"about_ca_system_score_gemma":0.0026771442,"threshold_uncertainty_score":0.026607215},"labels":[],"label_agreement":null},{"id":"W4402353118","doi":"10.1109/cisce62493.2024.10653102","title":"Exploration of the Possibility for a Wider Range of the Discount Factor for Reinforcement Learning","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Reinforcement learning; Reinforcement; Range (aeronautics); Computer science; Factor (programming language); Artificial intelligence; Machine learning; Psychology; Engineering; Social psychology; Aerospace engineering; Programming language","score_opus":0.0494761660167853,"score_gpt":0.2979828257252523,"score_spread":0.24850665970846703,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402353118","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20058723,0.0049855304,0.77613544,0.0025167263,0.00022441798,0.00026608544,0.00018616173,0.00072103966,0.014377363],"genre_scores_gemma":[0.8898701,0.00096057623,0.10797331,0.000264305,0.00005342599,0.00015346055,0.000057463418,0.00009744123,0.0005698715],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99688774,0.001510499,0.00021469676,0.0007596784,0.00040480937,0.00022251514],"domain_scores_gemma":[0.9532889,0.039356194,0.0019886396,0.0030914713,0.0012231806,0.0010515593],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008900735,0.0015712954,0.0012122004,0.0008873893,0.00056880433,0.002901633,0.0023158577,0.0015389622,0.0023738535],"category_scores_gemma":[0.055460036,0.0007810616,0.00094348117,0.0007167585,0.0017352245,0.0065871812,0.002143907,0.0045049805,0.00031442897],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016931287,0.0009313079,0.009369234,0.0009450649,0.000334312,0.0004455341,0.0008564805,0.6807818,0.012131328,0.1360342,0.0020657969,0.1544118],"study_design_scores_gemma":[0.000298273,0.00092659,0.0016876421,0.000397118,0.00014508038,0.0002730706,0.0002069279,0.84564066,0.005031927,0.13989341,0.0053678285,0.00013156877],"about_ca_topic_score_codex":0.0025029047,"about_ca_topic_score_gemma":0.0027477227,"teacher_disagreement_score":0.008900735,"about_ca_system_score_codex":0.001525302,"about_ca_system_score_gemma":0.0015031254,"threshold_uncertainty_score":0.047072113},"labels":[],"label_agreement":null},{"id":"W4402474348","doi":"10.1109/ccece59415.2024.10667224","title":"LLM4RL: Enhancing Reinforcement Learning with Large Language Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Reinforcement learning; Computer science; Reinforcement; Artificial intelligence; Human–computer interaction; Engineering","score_opus":0.010826491175834884,"score_gpt":0.24602232463981402,"score_spread":0.23519583346397913,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402474348","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011986908,0.00018230481,0.9820663,0.0002980897,0.00006909948,0.00006669326,0.000060498714,0.003744903,0.0015251847],"genre_scores_gemma":[0.59858876,0.00019157048,0.39609537,0.00050399586,0.000068071204,0.00026663794,0.0002749628,0.00048002382,0.0035306108],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99905235,0.0004050053,0.000047651632,0.00017193102,0.00023707887,0.000085986874],"domain_scores_gemma":[0.997521,0.0015007829,0.00016278854,0.0003530705,0.0003095247,0.00015284884],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020340637,0.00091631233,0.0008139384,0.00034133068,0.0003510303,0.00089837663,0.0020308131,0.0010685719,0.0035757138],"category_scores_gemma":[0.008145447,0.00043704125,0.00065244833,0.0002509625,0.0008531513,0.001734427,0.002609151,0.0023239804,0.0008983792],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021757203,0.00035436155,0.0013174161,0.00018306103,0.00009157447,0.00017998247,0.00017185975,0.77484846,0.01175163,0.017512308,0.005254717,0.18811707],"study_design_scores_gemma":[0.000014851936,0.00003095498,0.000026257021,0.00000399577,0.0000048560046,0.000008784254,0.0000041842936,0.9934676,0.0012107667,0.0045960625,0.00062694954,0.0000046266196],"about_ca_topic_score_codex":0.0044615427,"about_ca_topic_score_gemma":0.005555122,"teacher_disagreement_score":0.0044615427,"about_ca_system_score_codex":0.0007648355,"about_ca_system_score_gemma":0.0014891508,"threshold_uncertainty_score":0.011961937},"labels":[],"label_agreement":null},{"id":"W4402474374","doi":"10.1109/ccece59415.2024.10667230","title":"Continuous Action Learning Automata: A Strategy for Dynamic Optimization of Invariant Kalman Filter Covariances","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Kalman filter; Learning automata; Automaton; Computer science; Invariant (physics); Control theory (sociology); Extended Kalman filter; Mathematical optimization; Artificial intelligence; Mathematics; Control (management)","score_opus":0.03360847323668419,"score_gpt":0.2998974205379099,"score_spread":0.2662889473012257,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402474374","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010403216,0.000102431346,0.98688763,0.00013170896,0.000037204325,0.000025102936,0.000029724455,0.00042957673,0.0019533972],"genre_scores_gemma":[0.8759424,0.00014334077,0.12099213,0.0001285173,0.000048211456,0.00021594885,0.00007949544,0.000087663,0.002362195],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993048,0.00021288489,0.000048818,0.00019036692,0.00017564319,0.00006755735],"domain_scores_gemma":[0.99731284,0.0015959818,0.00025590585,0.00028615986,0.00042078464,0.00012844079],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013547539,0.0008282051,0.00082030677,0.0005490719,0.0005230201,0.0010732649,0.0014670821,0.0009950569,0.0019963954],"category_scores_gemma":[0.0052277874,0.00038601484,0.0005720407,0.0003736567,0.0015808407,0.0010422022,0.0015648626,0.0015039901,0.0003462429],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005827309,0.000043325534,0.0008562151,0.000051359064,0.000044615146,0.00008977405,0.00013200387,0.9114242,0.0026970012,0.03472104,0.0009439286,0.04893829],"study_design_scores_gemma":[0.0000054190255,0.00002116354,0.000036727106,0.000003930022,0.0000041812696,0.000008621467,0.000004299113,0.99360013,0.0003976976,0.0055950526,0.00031809177,0.0000046812725],"about_ca_topic_score_codex":0.0060204677,"about_ca_topic_score_gemma":0.005393375,"teacher_disagreement_score":0.0060204677,"about_ca_system_score_codex":0.00087685516,"about_ca_system_score_gemma":0.0010353581,"threshold_uncertainty_score":0.011970818},"labels":[],"label_agreement":null},{"id":"W4402550125","doi":"10.1098/rstb.2023.0416","title":"Elements of episodic memory: insights from artificial agents","year":2024,"lang":"en","type":"review","venue":"Philosophical Transactions of the Royal Society B Biological Sciences","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Institute for Advanced Research","funders":"British Academy; UK Research and Innovation","keywords":"Episodic memory; Cognitive science; Computer science; Psychology; Cognitive psychology; Neuroscience; Cognition","score_opus":0.14850100404059183,"score_gpt":0.3432534744909159,"score_spread":0.1947524704503241,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402550125","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00030013162,0.99327797,0.0016744724,0.0018979632,0.00016137965,0.000003018487,0.00000782089,0.000006662052,0.0026707104],"genre_scores_gemma":[0.007058023,0.9899256,0.0011546166,0.00074199465,0.00043660108,0.000011435158,0.00001696068,0.000003904801,0.00065084087],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9997336,0.000084750354,0.000026850605,0.000042322896,0.000091052374,0.000021396656],"domain_scores_gemma":[0.99871194,0.0009507983,0.00006261782,0.000050495622,0.00016833961,0.00005585663],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012665825,0.0005730901,0.0007697597,0.0023803532,0.00041571842,0.001721967,0.0011342225,0.001907379,0.0010012467],"category_scores_gemma":[0.002086963,0.0003200065,0.00033933914,0.0021439143,0.0035513868,0.0039511058,0.0010883341,0.0028204995,0.00053319783],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000056261597,0.000040291183,0.00041577558,0.0073774667,0.000079162135,0.00018553255,0.0006757736,0.0014937554,0.00060862507,0.19282416,0.01241195,0.7838313],"study_design_scores_gemma":[0.000012513168,0.00006245931,0.0011108983,0.0039507626,0.00004953474,0.00081151945,0.00030967908,0.0005511823,0.00045843594,0.121255,0.87139475,0.000033232813],"about_ca_topic_score_codex":0.0017032162,"about_ca_topic_score_gemma":0.0018818142,"teacher_disagreement_score":0.0023803532,"about_ca_system_score_codex":0.0014928387,"about_ca_system_score_gemma":0.0012846243,"threshold_uncertainty_score":0.010831356},"labels":[],"label_agreement":null},{"id":"W4402596039","doi":"10.21203/rs.3.rs-4314484/v1","title":"On-policy Actor-Critic Reinforcement Learning for Multi-UAV Exploration","year":2024,"lang":"en","type":"preprint","venue":"Research Square","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Regina","funders":"","keywords":"Reinforcement learning; Policy learning; Computer science; Reinforcement; Artificial intelligence; Psychology; Social psychology; Machine learning","score_opus":0.18191957642009787,"score_gpt":0.4539249079263171,"score_spread":0.27200533150621925,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402596039","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02785453,0.00041486247,0.96647793,0.00042645846,0.00011226702,0.00005172502,0.000045502606,0.00040148423,0.0042152624],"genre_scores_gemma":[0.9417367,0.0001632173,0.052984342,0.00013703674,0.00005209023,0.00011508461,0.000072658375,0.00009026516,0.004648584],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994997,0.00020411928,0.000021867028,0.00009064392,0.00009054189,0.00009304683],"domain_scores_gemma":[0.9980159,0.0013973557,0.00013945466,0.00010710475,0.00021267217,0.00012749217],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015393668,0.0010304919,0.0015060904,0.00040770622,0.0004727864,0.00079627,0.0011135621,0.0015893627,0.0033437123],"category_scores_gemma":[0.00508705,0.00062434206,0.00044204906,0.0003959721,0.0012428168,0.0009597995,0.0016601518,0.0018837505,0.00043490343],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008682521,0.000025954945,0.00019520923,0.000031098447,0.000017099312,0.00003083364,0.00002059247,0.9853705,0.0004018277,0.0036846015,0.0005398277,0.009595697],"study_design_scores_gemma":[0.000008033463,0.00000966949,0.000020295616,0.0000022904867,0.0000015987673,0.0000022643414,0.0000016513561,0.998355,0.00007674328,0.0014441084,0.00007710304,0.0000012461927],"about_ca_topic_score_codex":0.0071750986,"about_ca_topic_score_gemma":0.0049807616,"teacher_disagreement_score":0.0071750986,"about_ca_system_score_codex":0.0011884567,"about_ca_system_score_gemma":0.0013536756,"threshold_uncertainty_score":0.01426667},"labels":[],"label_agreement":null},{"id":"W4402754120","doi":"10.1109/cvpr52733.2024.00042","title":"Quantifying Task Priority for Multi-Task Optimization","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Defense Acquisition Program Administration; Agency for Defense Development; National Research Foundation of Korea","keywords":"Computer science; Task (project management); Task analysis; Human–computer interaction; Systems engineering; Engineering","score_opus":0.07975036434683974,"score_gpt":0.3384239884084495,"score_spread":0.25867362406160976,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402754120","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.048610736,0.00015214644,0.94863063,0.00014421387,0.00003479793,0.00009832056,0.000020131278,0.0003828311,0.0019261857],"genre_scores_gemma":[0.7872214,0.00010061484,0.21054281,0.00012909411,0.000027407876,0.00018597566,0.00005609943,0.00012693825,0.0016097478],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991014,0.00027689547,0.000052464686,0.00016033674,0.0002880243,0.00012091523],"domain_scores_gemma":[0.9978289,0.0009634699,0.0003233224,0.00030312416,0.00037074895,0.00021037646],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022287024,0.0014563543,0.00084300496,0.0007011743,0.00044684004,0.00081244286,0.0013960985,0.0011110493,0.001593699],"category_scores_gemma":[0.007187337,0.00045040532,0.00040518443,0.00042090227,0.00095325056,0.0021240565,0.0017088741,0.0017421825,0.0002870642],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024740509,0.00020267683,0.0016235423,0.0001231448,0.00006509953,0.00007057059,0.00009243107,0.86156565,0.018750235,0.011427601,0.00074587524,0.10508574],"study_design_scores_gemma":[0.0000112121,0.00006681888,0.00022818857,0.0000051105544,0.000006981411,0.000013778983,0.0000068100353,0.9908125,0.002525704,0.0061095115,0.00020707966,0.0000063318826],"about_ca_topic_score_codex":0.0012810883,"about_ca_topic_score_gemma":0.0016877869,"teacher_disagreement_score":0.0022287024,"about_ca_system_score_codex":0.0010377894,"about_ca_system_score_gemma":0.0014149427,"threshold_uncertainty_score":0.011786699},"labels":[],"label_agreement":null},{"id":"W4402891929","doi":"10.1109/tro.2024.3468770","title":"Using Implicit Behavior Cloning and Dynamic Movement Primitive to Facilitate Reinforcement Learning for Robot Motion Planning","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Robotics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Cloning (programming); Robot; Computer science; Motion (physics); Motion planning; Movement (music); Artificial intelligence; Physics; Programming language","score_opus":0.06835264497984908,"score_gpt":0.3177744257784029,"score_spread":0.24942178079855384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402891929","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05642244,0.0003113001,0.9372937,0.0002398524,0.000053750267,0.00016228203,0.000120622615,0.0033655264,0.0020305065],"genre_scores_gemma":[0.67809653,0.00016219163,0.31899622,0.00014016022,0.00001859849,0.0002380929,0.00027771553,0.00014576303,0.0019248035],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99951565,0.00017861913,0.000027720887,0.0001499266,0.000089173176,0.000038927243],"domain_scores_gemma":[0.99858546,0.00065999274,0.00018273426,0.00035769408,0.00012888758,0.00008525889],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010747889,0.0007801928,0.0005080984,0.00029285,0.00026632522,0.00037733527,0.0014181149,0.000667924,0.001853301],"category_scores_gemma":[0.003481584,0.00037301567,0.0004532598,0.00027368698,0.0010412689,0.00097360223,0.00089610595,0.0017470206,0.00047653064],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032049944,0.00047557827,0.004036067,0.00031978457,0.00007340805,0.00016635437,0.00019517227,0.61097014,0.03186423,0.010700451,0.0027111608,0.33816704],"study_design_scores_gemma":[0.000028775841,0.000105689884,0.00041966807,0.000007834011,0.000008837439,0.00002703212,0.000008412237,0.99070543,0.004759135,0.0027247122,0.0011954536,0.000008937775],"about_ca_topic_score_codex":0.004033713,"about_ca_topic_score_gemma":0.0056912876,"teacher_disagreement_score":0.004033713,"about_ca_system_score_codex":0.00056576997,"about_ca_system_score_gemma":0.0012347896,"threshold_uncertainty_score":0.008020461},"labels":[],"label_agreement":null},{"id":"W4403306447","doi":"10.1017/9781009302180.030","title":"Designing Dynamic Programming Algorithms via Reductions","year":2024,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Computer science; Dynamic programming; Algorithm","score_opus":0.020499562192775475,"score_gpt":0.217772213590013,"score_spread":0.19727265139723754,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403306447","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0015991372,0.001715034,0.91977286,0.0014482066,0.0003607442,0.00012348445,0.00023070944,0.0013206768,0.073429205],"genre_scores_gemma":[0.046830505,0.005939172,0.8715495,0.0011345708,0.00033379707,0.0008939912,0.001046892,0.0016540188,0.07061759],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9991204,0.00023197918,0.000059256232,0.00018197631,0.00034756731,0.000058752725],"domain_scores_gemma":[0.99944526,0.00034331184,0.000025620597,0.000091065274,0.0000764068,0.000018398716],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010742362,0.0012715493,0.0006895159,0.00077165564,0.00061712036,0.0023394667,0.0012073122,0.00072944036,0.023115242],"category_scores_gemma":[0.0038803497,0.00073701475,0.0011138248,0.0010721172,0.0016880487,0.003376786,0.0018841819,0.0036420017,0.010812995],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000247736,0.000036658606,0.00011080555,0.00043044894,0.000034298275,0.000053994074,0.0003056191,0.021591252,0.0018403981,0.71634716,0.03927353,0.21995108],"study_design_scores_gemma":[0.00002655449,0.000040141993,0.00012754423,0.00024981308,0.000016027583,0.000119416545,0.00007824891,0.039638847,0.002234084,0.69610775,0.2613358,0.00002578073],"about_ca_topic_score_codex":0.00078514277,"about_ca_topic_score_gemma":0.0008877289,"teacher_disagreement_score":0.023115242,"about_ca_system_score_codex":0.0012307216,"about_ca_system_score_gemma":0.0011020703,"threshold_uncertainty_score":0.077328205},"labels":[],"label_agreement":null},{"id":"W4403429602","doi":"10.1145/3638530.3654415","title":"Generational Information Transfer with Neuroevolution on Control Tasks","year":2024,"lang":"en","type":"article","venue":"Proceedings of the Genetic and Evolutionary Computation Conference Companion","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Neuroevolution; Computer science; Control (management); Artificial intelligence; Artificial neural network","score_opus":0.0115250681987799,"score_gpt":0.2069122778195792,"score_spread":0.19538720962079928,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403429602","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27422708,0.00028175226,0.706215,0.00060477614,0.00011685665,0.00022510921,0.00013482326,0.0023492447,0.015845414],"genre_scores_gemma":[0.840141,0.000111178786,0.15612046,0.00014389552,0.00002195035,0.00024195213,0.00011568155,0.0001407285,0.00296309],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995442,0.0001651755,0.000022810409,0.00007309815,0.00013898738,0.00005581741],"domain_scores_gemma":[0.99888295,0.0006138444,0.00007032248,0.0001889756,0.00019182965,0.000052064082],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009617035,0.0006350617,0.0006005593,0.00048333738,0.00046855243,0.0007187113,0.0012973252,0.00084194046,0.001785003],"category_scores_gemma":[0.004430813,0.00027908507,0.00043647853,0.000519101,0.0008268731,0.0008172725,0.0010371434,0.0011134688,0.00028169976],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005892518,0.00006843496,0.0007781897,0.000023279359,0.000024439265,0.00006762754,0.00008534854,0.9464493,0.0026897325,0.006571426,0.00075031794,0.042432956],"study_design_scores_gemma":[0.000014562301,0.00003104684,0.0001907284,0.0000035737503,0.0000060610296,0.000012614272,0.00000884953,0.9942194,0.0012343011,0.003714194,0.00055926165,0.000005486169],"about_ca_topic_score_codex":0.006316322,"about_ca_topic_score_gemma":0.0038533697,"teacher_disagreement_score":0.006316322,"about_ca_system_score_codex":0.0010462103,"about_ca_system_score_gemma":0.00078743894,"threshold_uncertainty_score":0.012559116},"labels":[],"label_agreement":null},{"id":"W4403534366","doi":"10.1109/codit62066.2024.10708505","title":"Combining Dense and Sparse Rewards to Improve Deep Reinforcement Learning Policies in Reach-Avoid Games with Faster Evaders in Two vs. One Scenarios","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Royal Military College of Canada; Queen's University","funders":"","keywords":"Reinforcement learning; Computer science; Reinforcement; Artificial intelligence; Human–computer interaction; Psychology; Social psychology","score_opus":0.01728873881557427,"score_gpt":0.26571061302666155,"score_spread":0.24842187421108727,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403534366","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2928787,0.00049637986,0.70035386,0.00056685903,0.00010527752,0.00010859173,0.000056346373,0.0007194454,0.0047145747],"genre_scores_gemma":[0.9722425,0.0000595317,0.026513398,0.000087795575,0.000014900977,0.000042630232,0.000024255369,0.000027835955,0.000987076],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99949217,0.00016930803,0.000027301563,0.00009520662,0.00010893633,0.00010706705],"domain_scores_gemma":[0.99762136,0.0014940647,0.00026810908,0.00014186955,0.00023429558,0.00024036376],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018094591,0.0012724773,0.0009642591,0.00043816774,0.00032388823,0.00066234136,0.001022909,0.0009350058,0.0012403239],"category_scores_gemma":[0.0066180006,0.00040116196,0.00032500117,0.00021368128,0.0009451222,0.00128553,0.0014747075,0.0016037363,0.00019692817],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011495497,0.00013130136,0.0011556322,0.000042745243,0.000027886575,0.00005244309,0.00004323905,0.9688956,0.0016253336,0.004619559,0.00032277577,0.022968424],"study_design_scores_gemma":[0.000012996398,0.00007481564,0.00010391571,0.0000044858702,0.0000055288915,0.000008681659,0.0000049725036,0.9971668,0.00035174302,0.0021564295,0.00010544078,0.0000042108386],"about_ca_topic_score_codex":0.0033268805,"about_ca_topic_score_gemma":0.0038833327,"teacher_disagreement_score":0.0033268805,"about_ca_system_score_codex":0.0009106064,"about_ca_system_score_gemma":0.0011876611,"threshold_uncertainty_score":0.009569466},"labels":[],"label_agreement":null},{"id":"W4403577538","doi":"10.48550/arxiv.2410.12062","title":"MFC-EQ: Mean-Field Control with Envelope Q-Learning for Moving Decentralized Agents in Formation","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Envelope (radar); Mean field theory; Field (mathematics); Control (management); Decentralised system; Physics; Computer science; Mathematics; Artificial intelligence; Telecommunications; Pure mathematics; Condensed matter physics","score_opus":0.06673068477514002,"score_gpt":0.20929395275910592,"score_spread":0.1425632679839659,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403577538","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024165405,0.00020382619,0.9731933,0.00029398847,0.000043111275,0.00004514844,0.00003476214,0.00026591835,0.001754559],"genre_scores_gemma":[0.9086525,0.00015526473,0.088777296,0.00020164697,0.000041542935,0.00013901616,0.00008582149,0.000049610968,0.0018973103],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994516,0.00019359983,0.000021478827,0.00012265403,0.00010349013,0.00010718602],"domain_scores_gemma":[0.9976452,0.00151087,0.00027413334,0.00017353219,0.00024536462,0.00015091297],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019578927,0.00085009297,0.0012788646,0.00042425466,0.00048685,0.000759735,0.0016549361,0.0014047595,0.0020047666],"category_scores_gemma":[0.0052472726,0.000422405,0.0005738657,0.00046606886,0.0014971045,0.0012510844,0.0017075309,0.0015559745,0.0002354398],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000028890456,0.000023542132,0.00024224957,0.000021323096,0.000012187156,0.000022093927,0.00002008577,0.9835866,0.00032080003,0.0064695943,0.00033133102,0.008921338],"study_design_scores_gemma":[0.000007918264,0.000018075458,0.000028627323,0.0000019308013,0.0000015284068,0.0000035516262,0.000002173819,0.9963924,0.0000673782,0.0033339935,0.00014057064,0.000001765712],"about_ca_topic_score_codex":0.006424525,"about_ca_topic_score_gemma":0.0042566946,"teacher_disagreement_score":0.006424525,"about_ca_system_score_codex":0.0010694982,"about_ca_system_score_gemma":0.001505532,"threshold_uncertainty_score":0.012774229},"labels":[],"label_agreement":null},{"id":"W4403702578","doi":"10.48550/arxiv.2409.10096","title":"Robust Reinforcement Learning with Dynamic Distortion Risk Measures","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Reinforcement; Distortion (music); Computer science; Artificial intelligence; Psychology; Social psychology; Telecommunications","score_opus":0.052434498711610086,"score_gpt":0.1774675049997577,"score_spread":0.12503300628814762,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403702578","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010413791,0.00013470573,0.9881287,0.00015712348,0.000014570088,0.00002551858,0.00001575907,0.00015829178,0.0009515381],"genre_scores_gemma":[0.84551406,0.0001941727,0.15089096,0.00014910428,0.00004037179,0.00014080621,0.0000745246,0.00010685301,0.002889102],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99861324,0.0006152391,0.00006626325,0.00030679643,0.00026066517,0.00013774514],"domain_scores_gemma":[0.9962823,0.002548228,0.00040933196,0.0002603913,0.0003449206,0.00015484024],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034614638,0.001367917,0.0016194816,0.00045567195,0.00033112138,0.0013446874,0.0014517287,0.0014429567,0.0015165255],"category_scores_gemma":[0.010249922,0.00062076567,0.0007277327,0.0004319425,0.0015801055,0.0018453836,0.0017045374,0.0021679767,0.00028669141],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000038796617,0.000023723835,0.00030845302,0.000025667709,0.000028405984,0.00003759397,0.00002914769,0.9707925,0.00041827484,0.017497333,0.00027814688,0.010522024],"study_design_scores_gemma":[0.0000071883696,0.000014268826,0.00003116589,0.0000030293404,0.0000030624226,0.000006040044,0.0000020178682,0.9911986,0.00014868756,0.008487278,0.000095358984,0.0000032583343],"about_ca_topic_score_codex":0.003945538,"about_ca_topic_score_gemma":0.0017768355,"teacher_disagreement_score":0.003945538,"about_ca_system_score_codex":0.0015453552,"about_ca_system_score_gemma":0.001330946,"threshold_uncertainty_score":0.018306196},"labels":[],"label_agreement":null},{"id":"W4403760022","doi":"10.1016/j.brachy.2024.08.113","title":"PSOR03 Presentation Time: 11:40 AM","year":2024,"lang":"en","type":"article","venue":"Brachytherapy","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval; Centre hospitalier de l'Université Laval; Centre hospitalier universitaire de Québec","funders":"","keywords":"Medicine; Presentation (obstetrics); Surgery","score_opus":0.012872976264696955,"score_gpt":0.2754509554621671,"score_spread":0.2625779791974701,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403760022","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0015320096,0.00035185227,0.003748866,0.0019026306,0.006899571,0.00044806005,0.0043209223,0.010616785,0.9701793],"genre_scores_gemma":[0.0045723775,0.00011875175,0.00044711147,0.00032941924,0.00064545247,0.00009847589,0.0011060286,0.0017867947,0.9908957],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9993988,0.00006728022,0.000018450977,0.00012511881,0.00021755842,0.000172803],"domain_scores_gemma":[0.9982982,0.00019978237,0.000055363344,0.00016488734,0.00053978135,0.0007420034],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0011041473,0.001884215,0.0020396865,0.0012488521,0.0020521227,0.007303732,0.0019352003,0.004261291,0.9487048],"category_scores_gemma":[0.0026866267,0.00071941264,0.0019559257,0.0008422185,0.00069454405,0.0014122294,0.0035128768,0.0026137487,0.89306045],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00068232906,0.00017987727,0.00010191437,0.00022908415,0.000019207706,0.000110574256,0.000060063798,0.00026663864,0.0029455947,0.0022669558,0.93493897,0.058198888],"study_design_scores_gemma":[0.00016630258,0.0002931564,0.0011264047,0.00019862456,0.000019903404,0.00006660975,0.00007446809,0.0013004892,0.0014159863,0.0017344701,0.99357444,0.000029093953],"about_ca_topic_score_codex":0.0022472218,"about_ca_topic_score_gemma":0.004754706,"teacher_disagreement_score":0.05129522,"about_ca_system_score_codex":0.001339963,"about_ca_system_score_gemma":0.0011999221,"threshold_uncertainty_score":0.07316643},"labels":[],"label_agreement":null},{"id":"W4403793598","doi":"10.1007/978-3-031-75872-0_10","title":"Model-Driven Design and Generation of Training Simulators for Reinforcement Learning","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Regina; University of Toronto; York University","funders":"","keywords":"Computer science; Reinforcement learning; Training (meteorology); Artificial intelligence; Human–computer interaction; Simulation","score_opus":0.07555659613702641,"score_gpt":0.27975780820531954,"score_spread":0.20420121206829311,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403793598","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0072666127,0.00008466026,0.98981386,0.000063773776,0.000024277184,0.000080148304,0.000035395195,0.0004863761,0.0021448971],"genre_scores_gemma":[0.6295735,0.00014940799,0.36592135,0.000081451486,0.00001840557,0.00048850133,0.00014626647,0.00022402141,0.0033970599],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997199,0.00008696117,0.000012960842,0.000056346595,0.0000876179,0.000036143625],"domain_scores_gemma":[0.99902666,0.0005940907,0.000075707496,0.00008182062,0.00018506171,0.000036655238],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007550944,0.00066429854,0.0006562005,0.00037235982,0.00029282013,0.00061003654,0.001388641,0.0010360408,0.003732745],"category_scores_gemma":[0.0027002965,0.00060793303,0.0006312576,0.00024011414,0.0005884831,0.0004447541,0.0008528323,0.0010677218,0.00056082086],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000030287552,0.000028872128,0.00012621487,0.000042002044,0.000010041795,0.000025567972,0.000026108359,0.96942836,0.0021322805,0.0069750915,0.00045915076,0.020715946],"study_design_scores_gemma":[0.00000455688,0.000010037845,0.000015445614,0.000002649526,0.0000020175648,0.000005531163,0.0000015394914,0.99765635,0.0005073812,0.0015527019,0.00024037196,0.0000013507699],"about_ca_topic_score_codex":0.0027057426,"about_ca_topic_score_gemma":0.003282384,"teacher_disagreement_score":0.003732745,"about_ca_system_score_codex":0.0008121212,"about_ca_system_score_gemma":0.0011244601,"threshold_uncertainty_score":0.012487233},"labels":[],"label_agreement":null},{"id":"W4403906488","doi":"10.1007/978-3-031-73033-7_9","title":"Learning to Drive via Asymmetric Self-Play","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Human–computer interaction; Artificial intelligence","score_opus":0.00983572967448916,"score_gpt":0.23941293751926895,"score_spread":0.2295772078447798,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403906488","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09600129,0.00028959784,0.8482579,0.00041992863,0.00013710496,0.00006911166,0.000061196435,0.00048951386,0.054274414],"genre_scores_gemma":[0.9515278,0.00016409485,0.024391761,0.00007283984,0.000034164423,0.000086642365,0.000046723646,0.00004884889,0.023627246],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99988854,0.000025539897,0.000006692755,0.000027261905,0.000030865576,0.000021101212],"domain_scores_gemma":[0.999622,0.00022321015,0.000035509005,0.0000425217,0.000035403642,0.000041348518],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00027479872,0.00045713724,0.00029833138,0.00016352395,0.0002444222,0.00050241215,0.00062191155,0.00043313595,0.006726088],"category_scores_gemma":[0.0013710792,0.00020150174,0.00023755779,0.00012974534,0.00061814935,0.0008396079,0.0011316264,0.00089267426,0.00074389216],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003331632,0.00019898277,0.0009251895,0.00018538401,0.00005751421,0.00015986474,0.00023855592,0.45947403,0.018076487,0.27201095,0.005116833,0.24322301],"study_design_scores_gemma":[0.000015300227,0.000089895526,0.00015219365,0.000011173183,0.000005669367,0.000058212037,0.000018855224,0.9288342,0.0017346427,0.0672896,0.0017834251,0.000006897885],"about_ca_topic_score_codex":0.00055080664,"about_ca_topic_score_gemma":0.00066400075,"teacher_disagreement_score":0.006726088,"about_ca_system_score_codex":0.0002637297,"about_ca_system_score_gemma":0.00027696314,"threshold_uncertainty_score":0.022500992},"labels":[],"label_agreement":null},{"id":"W4403935894","doi":"10.1145/3652620.3676878","title":"Towards Model Repair by Human Opinion--Guided Reinforcement Learning","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Reinforcement learning; Computer science; Reinforcement; Artificial intelligence; Human–computer interaction; Engineering; Structural engineering","score_opus":0.037632307705128265,"score_gpt":0.3118472252698286,"score_spread":0.27421491756470034,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403935894","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041378308,0.00016677858,0.95563966,0.0003393627,0.00003344787,0.000046761248,0.00003304076,0.0005909911,0.0017715856],"genre_scores_gemma":[0.86976045,0.00009122123,0.12842204,0.0002297027,0.00004641921,0.000100136866,0.0001025663,0.00010038872,0.0011469991],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99908864,0.000386468,0.000037798505,0.00023082999,0.00015904995,0.0000972557],"domain_scores_gemma":[0.9946679,0.0036119735,0.00060764246,0.00040227512,0.0004604449,0.00024973767],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020382768,0.001157165,0.0010131071,0.0005555305,0.00038308906,0.00090371585,0.0015004941,0.0013659683,0.0018710018],"category_scores_gemma":[0.010814928,0.00043334244,0.00061249186,0.0003114871,0.0010752113,0.0012741382,0.0015952924,0.0017230345,0.00041106206],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015119839,0.00013949201,0.002686755,0.00012651471,0.00006844746,0.00013223397,0.00035358887,0.89421797,0.0038610802,0.010683607,0.0018493697,0.085729785],"study_design_scores_gemma":[0.00001037147,0.000021727006,0.000064145854,0.0000049064715,0.0000049502905,0.000008896572,0.000012358619,0.994783,0.00034093022,0.004544921,0.00019999819,0.0000037769453],"about_ca_topic_score_codex":0.004364211,"about_ca_topic_score_gemma":0.004416667,"teacher_disagreement_score":0.004364211,"about_ca_system_score_codex":0.00081100373,"about_ca_system_score_gemma":0.0011978186,"threshold_uncertainty_score":0.01077956},"labels":[],"label_agreement":null},{"id":"W4404299250","doi":"10.2139/ssrn.4978973","title":"Ill-Posedness Evolved in the Deep: Adaptive Evolutionary Latent Optimization, An Application to Recovering Volatility","year":2024,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Adaptive evolution; Volatility (finance); Econometrics; Economics; Mathematical optimization; Computer science; Artificial intelligence; Mathematics; Biology","score_opus":0.014087244054025633,"score_gpt":0.25577106849866643,"score_spread":0.2416838244446408,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404299250","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.087020226,0.0002346964,0.9096399,0.0009173342,0.000050681923,0.000020608197,0.00005749075,0.00016168288,0.0018974519],"genre_scores_gemma":[0.85249674,0.00019696847,0.14070858,0.00032881828,0.000067534056,0.00008269506,0.00009106891,0.00018681744,0.0058406824],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99976474,0.00010516344,0.00001157488,0.000043627857,0.000044524117,0.000030332103],"domain_scores_gemma":[0.997715,0.0017320398,0.00016195976,0.00011760528,0.0001613625,0.00011202193],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017425192,0.0005998874,0.0009717599,0.0003389203,0.00036779864,0.0010580261,0.0009978847,0.0021448946,0.0019344838],"category_scores_gemma":[0.0069964365,0.0005764873,0.0006267226,0.00041512577,0.0016249237,0.0017728473,0.0026805983,0.0023804065,0.00014714456],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006311033,0.000045576646,0.0006852918,0.00005198458,0.000036853944,0.00005996367,0.000073950694,0.9513605,0.0020317268,0.03335855,0.000602236,0.011630245],"study_design_scores_gemma":[0.000003468677,0.000005519331,0.000032722182,0.0000017982119,0.0000016652596,0.0000034983436,0.0000023973448,0.99485403,0.00013175084,0.0049234186,0.000037500748,0.0000022498818],"about_ca_topic_score_codex":0.002645866,"about_ca_topic_score_gemma":0.0030251155,"teacher_disagreement_score":0.002645866,"about_ca_system_score_codex":0.0007858598,"about_ca_system_score_gemma":0.00093256,"threshold_uncertainty_score":0.0092154145},"labels":[],"label_agreement":null},{"id":"W4404317069","doi":"10.1109/tse.2024.3470368","title":"Scoping Software Engineering for AI: The TSE Perspective","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa; University of Alberta; Queen's University; University of Toronto","funders":"","keywords":"Computer science; Software engineering; Perspective (graphical); Software development; Software; Programming language; Artificial intelligence","score_opus":0.013694603385929241,"score_gpt":0.25763863923195046,"score_spread":0.24394403584602123,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404317069","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0028931568,0.12031658,0.119155094,0.55472887,0.092927665,0.00025264345,0.00015201473,0.00032235906,0.109251596],"genre_scores_gemma":[0.21388745,0.23615925,0.109763086,0.13464603,0.19526556,0.0011674277,0.0005350393,0.0011861911,0.107389964],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9670784,0.02037909,0.0035797327,0.001912385,0.005910221,0.0011402515],"domain_scores_gemma":[0.8202537,0.13368438,0.007833165,0.009474794,0.02342381,0.005330106],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.055379555,0.000877978,0.0011264607,0.008254416,0.0040809126,0.01923783,0.0024299994,0.009804505,0.0069432626],"category_scores_gemma":[0.13607809,0.00065284636,0.0011607402,0.0066265985,0.02087826,0.01681613,0.008259543,0.009401423,0.0025638663],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000023995533,0.000021085401,0.00022177865,0.0015504557,0.00003288197,0.00038081236,0.0041159145,0.00066420774,0.00019873209,0.8297703,0.086215824,0.07680396],"study_design_scores_gemma":[0.000013388757,0.000024343834,0.000116101386,0.003630896,0.000018027471,0.00033043843,0.0021019878,0.000719367,0.00021154978,0.4656753,0.5271363,0.000022237617],"about_ca_topic_score_codex":0.0012569071,"about_ca_topic_score_gemma":0.0018978943,"teacher_disagreement_score":0.055379555,"about_ca_system_score_codex":0.0043051913,"about_ca_system_score_gemma":0.013237346,"threshold_uncertainty_score":0.29287857},"labels":[],"label_agreement":null},{"id":"W4404535976","doi":"10.3390/e26110995","title":"Information-Theoretic Generalization Bounds for Batch Reinforcement Learning","year":2024,"lang":"en","type":"article","venue":"Entropy","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Generalization; Reinforcement learning; Computer science; Information theory; Generalization error; Artificial intelligence; Prior information; Machine learning; Mathematics; Mathematical optimization; Unsupervised learning; Statistics","score_opus":0.008531388615068268,"score_gpt":0.24575923156269766,"score_spread":0.23722784294762939,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404535976","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014974975,0.0009377733,0.97679013,0.0010086317,0.0000824665,0.00006638222,0.00013215715,0.00027250266,0.0057349326],"genre_scores_gemma":[0.8397178,0.0019998583,0.14876819,0.0010750621,0.0005775914,0.00063600927,0.00044737788,0.0004936172,0.0062843715],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9944312,0.0018693984,0.00031935127,0.001076001,0.0016797268,0.00062430196],"domain_scores_gemma":[0.9458694,0.042490132,0.0031374532,0.0042913863,0.0030226146,0.0011890141],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0116711585,0.0022394313,0.0023022746,0.0016996436,0.0010033016,0.0023396425,0.003815342,0.0022902873,0.005145691],"category_scores_gemma":[0.05326865,0.0008622993,0.0020984095,0.001548141,0.0047121122,0.008203965,0.0059107663,0.0074952277,0.00080831005],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022789955,0.00012059031,0.0012744615,0.00033622867,0.0001794564,0.00014181422,0.00028485054,0.5487391,0.0035750691,0.40768,0.0028473958,0.034593128],"study_design_scores_gemma":[0.000008513674,0.000055044562,0.00024602556,0.00003465887,0.000021429356,0.000036646048,0.000013363728,0.832119,0.00079062366,0.1661476,0.0005056156,0.000021464453],"about_ca_topic_score_codex":0.0028200021,"about_ca_topic_score_gemma":0.0020957314,"teacher_disagreement_score":0.0116711585,"about_ca_system_score_codex":0.0049617975,"about_ca_system_score_gemma":0.0017998719,"threshold_uncertainty_score":0.06172371},"labels":[],"label_agreement":null},{"id":"W4404688444","doi":"10.1109/tnnls.2024.3496492","title":"GenSafe: A Generalizable Safety Enhancer for Safe Reinforcement Learning Algorithms Based on Reduced Order Markov Decision Process Model","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Neural Networks and Learning Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"JST-Mirai Program; Natural Sciences and Engineering Research Council of Canada; Grant Foundation","keywords":"Markov decision process; Reinforcement learning; Computer science; Process (computing); Order (exchange); Machine learning; Markov process; Markov chain; Algorithm; Artificial intelligence; Mathematics; Business; Programming language; Statistics","score_opus":0.017102101971001402,"score_gpt":0.2699021736160693,"score_spread":0.25280007164506785,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404688444","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0064033493,0.000119718876,0.9914062,0.000119312994,0.0000228024,0.000051952324,0.000036249097,0.00065924553,0.0011811283],"genre_scores_gemma":[0.672821,0.00033251836,0.3219301,0.00044226935,0.000060754162,0.00037986692,0.000275766,0.00034746714,0.003410399],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99912137,0.00026093092,0.000046491463,0.00016580036,0.00030064574,0.000104856896],"domain_scores_gemma":[0.9978098,0.0012094073,0.00025378252,0.00024165989,0.00036508768,0.00012028951],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021642735,0.0011410873,0.0010416211,0.00049235264,0.00036824003,0.0007727322,0.0016331286,0.0010380895,0.0028357182],"category_scores_gemma":[0.0052070655,0.00050243415,0.0007451384,0.0002716895,0.0014469167,0.001141961,0.0021902754,0.002885742,0.0005471162],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013016962,0.000086478285,0.0010702296,0.00015572965,0.000047589034,0.00010021339,0.0001143624,0.9008088,0.0040353714,0.024127187,0.0012721787,0.06805177],"study_design_scores_gemma":[0.000014716272,0.0000515641,0.000046169247,0.0000086752425,0.000005928511,0.000013745044,0.0000036651193,0.9929825,0.0007243572,0.005600803,0.0005431767,0.000004750912],"about_ca_topic_score_codex":0.0027811762,"about_ca_topic_score_gemma":0.003386497,"teacher_disagreement_score":0.0028357182,"about_ca_system_score_codex":0.0008927947,"about_ca_system_score_gemma":0.0026509648,"threshold_uncertainty_score":0.01144594},"labels":[],"label_agreement":null},{"id":"W4404815779","doi":"10.1093/pnasnexus/pgae540","title":"Modeling long-term nutritional behaviors using deep homeostatic reinforcement learning","year":2024,"lang":"en","type":"article","venue":"PNAS Nexus","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Japan Society for the Promotion of Science; Japan Society for the Promotion of Science London; Japan Agency for Medical Research and Development","keywords":"Term (time); Reinforcement; Reinforcement learning; Psychology; Cognitive psychology; Computer science; Artificial intelligence; Social psychology; Physics","score_opus":0.034082258426414075,"score_gpt":0.2961966146611672,"score_spread":0.2621143562347531,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404815779","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4064566,0.00035799018,0.5845053,0.00054292066,0.000058343925,0.00004960792,0.00010207133,0.00041525805,0.007511937],"genre_scores_gemma":[0.987021,0.000060232116,0.011119537,0.000045292567,0.000007338481,0.000039869963,0.000029912646,0.000014140935,0.0016627354],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999897,0.00002775922,0.0000045118395,0.000026532152,0.000018192984,0.000025977171],"domain_scores_gemma":[0.99969935,0.00012122667,0.00007457641,0.000020624568,0.00005184838,0.000032329623],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00040710616,0.00042176966,0.00036839803,0.0002077747,0.00019875285,0.00050306914,0.0008101028,0.0006113412,0.0012149538],"category_scores_gemma":[0.0011200946,0.0002542885,0.00030735484,0.00014419055,0.00069352676,0.0005824735,0.0005792842,0.00064589723,0.00012222283],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000019225085,0.000022935474,0.0010368278,0.000009455525,0.000015721524,0.00002668533,0.000015407799,0.9913018,0.00087392377,0.002996785,0.00012635402,0.0035548373],"study_design_scores_gemma":[0.0000014460679,0.0000037278846,0.00006954289,5.280946e-7,0.0000010686845,0.0000015216616,0.0000012639597,0.9990778,0.00004979423,0.0007655877,0.000026900427,7.7934413e-7],"about_ca_topic_score_codex":0.006674022,"about_ca_topic_score_gemma":0.006421089,"teacher_disagreement_score":0.006674022,"about_ca_system_score_codex":0.0007182919,"about_ca_system_score_gemma":0.0006427486,"threshold_uncertainty_score":0.013270378},"labels":[],"label_agreement":null},{"id":"W4404864837","doi":"10.1016/j.iot.2024.101447","title":"Adaptive target localization under uncertainty using Multi-Agent Deep Reinforcement Learning with knowledge transfer","year":2024,"lang":"en","type":"article","venue":"Internet of Things","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada; Fonds de recherche du Québec – Nature et technologies; Alliance de recherche numérique du Canada","keywords":"Reinforcement learning; Artificial intelligence; Computer science; Transfer of learning; Knowledge transfer; Machine learning; Reinforcement; Knowledge management; Psychology; Social psychology","score_opus":0.03181315952003095,"score_gpt":0.27215590146951923,"score_spread":0.2403427419494883,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404864837","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037709277,0.00033438797,0.95852214,0.0002834141,0.000063472966,0.000036674566,0.000027290793,0.00062666903,0.0023966455],"genre_scores_gemma":[0.9588689,0.000117529664,0.0390362,0.00013687315,0.000029438488,0.000073507166,0.000050651957,0.000037049776,0.001649821],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996339,0.000086710155,0.000018887009,0.0000995765,0.00008974659,0.00007120131],"domain_scores_gemma":[0.9988833,0.00059692014,0.00018610673,0.00007907997,0.00017731759,0.00007727697],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009614221,0.0011122746,0.0010265925,0.00034710398,0.0003934306,0.00075465045,0.0013489204,0.0010541315,0.0010735793],"category_scores_gemma":[0.0025067744,0.00048543615,0.00059435726,0.00027783433,0.00091724796,0.0009607854,0.0015040039,0.0016644686,0.00019765581],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000033697477,0.000032827687,0.0004333071,0.00002556267,0.00002769926,0.00006325077,0.00003512793,0.97975487,0.0009952666,0.0019870766,0.00030578248,0.016305584],"study_design_scores_gemma":[0.0000033484173,0.000010383182,0.000025198296,0.0000015160156,0.0000023246403,0.0000037091218,0.0000019786717,0.9990207,0.00014072133,0.0007315113,0.000056941255,0.00000155625],"about_ca_topic_score_codex":0.008076503,"about_ca_topic_score_gemma":0.0052060355,"teacher_disagreement_score":0.008076503,"about_ca_system_score_codex":0.0009406242,"about_ca_system_score_gemma":0.0011357929,"threshold_uncertainty_score":0.016058981},"labels":[],"label_agreement":null},{"id":"W4404954737","doi":"10.1109/ssrr62954.2024.10770028","title":"Monte Carlo Tree Search for Behavior Planning in Autonomous Driving","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Monte Carlo tree search; Monte Carlo method; Computer science; Tree (set theory); Artificial intelligence; Mathematics; Statistics","score_opus":0.039345796654000696,"score_gpt":0.3198587520506013,"score_spread":0.28051295539660065,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404954737","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03015642,0.00034242895,0.96589035,0.00022378984,0.00003161493,0.000050540893,0.000041621566,0.0003465768,0.0029167663],"genre_scores_gemma":[0.73105043,0.00023541071,0.26637337,0.00013587261,0.00003318334,0.00020805119,0.00014806345,0.00009126562,0.001724347],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99962187,0.00016598374,0.000017481769,0.00005378579,0.00009841662,0.000042347518],"domain_scores_gemma":[0.99829346,0.0013110421,0.00009725988,0.000053223113,0.0001757664,0.00006931534],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010493957,0.00048401652,0.00065100886,0.000697011,0.00046144595,0.0006170898,0.00080624456,0.0008259593,0.0014690231],"category_scores_gemma":[0.0047981944,0.00040415174,0.0004489237,0.00063420646,0.0009159859,0.000718612,0.00068539206,0.0008236074,0.00025925462],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000025241954,0.000012591553,0.0004500762,0.000018784363,0.000011032358,0.000014819265,0.000024704537,0.9790162,0.00031434395,0.007852055,0.0002545511,0.012005764],"study_design_scores_gemma":[0.0000026256037,0.000005169068,0.000026360172,0.0000019234844,0.0000010662276,0.0000024969913,0.0000018545909,0.9978728,0.000059294667,0.001914719,0.00011045865,0.0000011471376],"about_ca_topic_score_codex":0.009830876,"about_ca_topic_score_gemma":0.009417708,"teacher_disagreement_score":0.009830876,"about_ca_system_score_codex":0.0011426476,"about_ca_system_score_gemma":0.0017289435,"threshold_uncertainty_score":0.019547284},"labels":[],"label_agreement":null},{"id":"W4405568984","doi":"10.1007/978-3-031-77688-5_42","title":"Leveraging Single and Multi-task Reinforcement Learning Algorithms for Autonomous Mobile Aloha Robot","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in networks and systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Aloha; Reinforcement learning; Computer science; Task (project management); Mobile robot; Artificial intelligence; Robot; Human–computer interaction; Throughput; Wireless; Engineering; Telecommunications","score_opus":0.02962780433733567,"score_gpt":0.250649203937621,"score_spread":0.22102139960028536,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405568984","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.050586198,0.00056869246,0.9430072,0.00015378062,0.00010964348,0.00003748033,0.000015466154,0.000793288,0.0047282623],"genre_scores_gemma":[0.9174917,0.00014486446,0.07837908,0.00006043568,0.00003714574,0.00005545477,0.000027092354,0.000045445377,0.003758697],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9998419,0.000031446354,0.000009059651,0.00003955961,0.000042232652,0.00003584334],"domain_scores_gemma":[0.9997008,0.00013501616,0.00002956544,0.000039118462,0.000069727816,0.000025808298],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00046173678,0.00046965852,0.0005463231,0.00022927858,0.00038498896,0.0004681951,0.0008566191,0.00061892247,0.0014988789],"category_scores_gemma":[0.0009189173,0.00025122246,0.00031099125,0.00021382555,0.00044918782,0.00059381797,0.0009286746,0.00073807745,0.00039523953],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013437275,0.00008156986,0.0005061656,0.00006751443,0.000045890178,0.00008209131,0.000060917824,0.8164381,0.011023978,0.00550359,0.0011964983,0.16485931],"study_design_scores_gemma":[0.0000043763844,0.000038436796,0.00006972761,0.0000021620442,0.000004068919,0.000013383015,0.000003942897,0.997186,0.0006542521,0.0017364142,0.00028409268,0.0000032300775],"about_ca_topic_score_codex":0.0029202143,"about_ca_topic_score_gemma":0.002796272,"teacher_disagreement_score":0.0029202143,"about_ca_system_score_codex":0.000325582,"about_ca_system_score_gemma":0.0005400141,"threshold_uncertainty_score":0.005806446},"labels":[],"label_agreement":null},{"id":"W4405602545","doi":"10.1109/icsme58944.2024.00019","title":"Toward Debugging Deep Reinforcement Learning Programs with RLExplorer","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Debugging; Reinforcement learning; Computer science; Algorithmic program debugging; Programming language; Artificial intelligence; Human–computer interaction; Software engineering","score_opus":0.030093056225010118,"score_gpt":0.24657515407600766,"score_spread":0.21648209785099753,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405602545","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027161784,0.00010740552,0.9595642,0.0002995758,0.000068813184,0.00004316177,0.00008391796,0.010931292,0.0017397288],"genre_scores_gemma":[0.4776283,0.000059583537,0.51865226,0.00028022644,0.000023149416,0.00007011149,0.00016726402,0.000994714,0.0021244478],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99879324,0.00052763085,0.000065757,0.00027273875,0.00023170492,0.00010891924],"domain_scores_gemma":[0.9938287,0.004465814,0.00025023,0.0008342074,0.00047042052,0.00015058473],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002295909,0.0011259243,0.00061744486,0.00046956787,0.00038056925,0.00095148664,0.0019782467,0.0013946288,0.005615567],"category_scores_gemma":[0.013535739,0.00089775195,0.00059373956,0.00026235086,0.0010772275,0.0024637475,0.0017783483,0.0027771138,0.00080756977],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009907348,0.00036748443,0.0045739627,0.0002933051,0.00012372363,0.00030900998,0.00029515394,0.5746325,0.01234392,0.036776382,0.008182513,0.36111128],"study_design_scores_gemma":[0.00002488215,0.00002458626,0.00006000422,0.000012485832,0.000009657195,0.000017215154,0.000013455954,0.98491573,0.0040185032,0.0102810515,0.00061797764,0.0000044248873],"about_ca_topic_score_codex":0.0033647015,"about_ca_topic_score_gemma":0.007979429,"teacher_disagreement_score":0.005615567,"about_ca_system_score_codex":0.00076860015,"about_ca_system_score_gemma":0.0013242572,"threshold_uncertainty_score":0.018785954},"labels":[],"label_agreement":null},{"id":"W4405778651","doi":"10.1109/iros58592.2024.10802721","title":"Guiding Reinforcement Learning with Incomplete System Dynamics","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Honeywell (Canada); University of British Columbia","funders":"","keywords":"Reinforcement learning; Computer science; Dynamics (music); Artificial intelligence; Human–computer interaction; Psychology","score_opus":0.018374155161227892,"score_gpt":0.22859966066940193,"score_spread":0.21022550550817404,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405778651","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02795012,0.000076711854,0.9705117,0.000104238214,0.0000162433,0.000026282783,0.000015916097,0.0004223319,0.0008763766],"genre_scores_gemma":[0.9335965,0.00006229444,0.06489045,0.00007612412,0.000020176423,0.000083232946,0.00003854362,0.000056266963,0.0011764909],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996307,0.00012265741,0.00001900796,0.00008466796,0.000085845924,0.000057153004],"domain_scores_gemma":[0.9978739,0.0014367092,0.0002304146,0.00017733904,0.00018803094,0.00009356008],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012596798,0.00089988793,0.00089082617,0.00029966893,0.00031462958,0.00053927395,0.00088806555,0.0006585458,0.0010011299],"category_scores_gemma":[0.004312738,0.00046912872,0.00032176738,0.00017404891,0.0010769643,0.0008762794,0.0010062213,0.0012692,0.00018308373],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000041672505,0.00002541597,0.00024998016,0.000030360765,0.000015211358,0.000031826,0.000037226255,0.98054194,0.0012781061,0.0035762773,0.00022149469,0.013950511],"study_design_scores_gemma":[0.0000061169094,0.0000122636475,0.0000231643,0.0000017620866,0.0000019584556,0.0000035415871,0.0000015765971,0.99812514,0.00027416975,0.0014684455,0.00008027268,0.0000015348907],"about_ca_topic_score_codex":0.005139455,"about_ca_topic_score_gemma":0.004308962,"teacher_disagreement_score":0.005139455,"about_ca_system_score_codex":0.00074198615,"about_ca_system_score_gemma":0.0013594804,"threshold_uncertainty_score":0.010219097},"labels":[],"label_agreement":null},{"id":"W4405787662","doi":"10.1109/iros58592.2024.10801347","title":"PGA: Personalizing Grasping Agents with Single Human-Robot Interaction","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Human–robot interaction; Robot; Human–computer interaction; Artificial intelligence","score_opus":0.06464383271075191,"score_gpt":0.31186956567148877,"score_spread":0.24722573296073685,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405787662","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0599973,0.0010574982,0.9052939,0.00037676241,0.00016352342,0.00054970325,0.0011270554,0.026682077,0.004752274],"genre_scores_gemma":[0.27081266,0.00044837556,0.71235704,0.00058405154,0.00006962143,0.00072457304,0.003902121,0.0013043572,0.009797235],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9985897,0.000285253,0.00006648342,0.0007159601,0.00023383315,0.0001087895],"domain_scores_gemma":[0.9984457,0.0004897939,0.00018866333,0.0005960384,0.00016153489,0.00011831419],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014357263,0.0026916105,0.0013072038,0.00097312685,0.00078792305,0.0011732723,0.0036560819,0.002312316,0.004883795],"category_scores_gemma":[0.0038430293,0.0009846025,0.0018202689,0.0006233984,0.0014074341,0.0029144913,0.0035588734,0.002103497,0.003545447],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00079517375,0.00081968907,0.005944661,0.0010355256,0.0003141738,0.00067610195,0.0007510733,0.17840432,0.056790765,0.0054225614,0.021847792,0.7271982],"study_design_scores_gemma":[0.00008295256,0.0005219409,0.002319247,0.00009143354,0.000066781,0.0004341598,0.00024729728,0.9433651,0.024226125,0.010253376,0.018304462,0.000087145454],"about_ca_topic_score_codex":0.0043836124,"about_ca_topic_score_gemma":0.009238816,"teacher_disagreement_score":0.004883795,"about_ca_system_score_codex":0.000881825,"about_ca_system_score_gemma":0.0015899718,"threshold_uncertainty_score":0.016337931},"labels":[],"label_agreement":null},{"id":"W4405938060","doi":"10.1109/icdici62993.2024.10810955","title":"Reinforcement Learning in Cognitive Robots for Autonomous Path Planning","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Horizon College and Seminary","funders":"","keywords":"Reinforcement learning; Computer science; Motion planning; Robot; Path (computing); Mobile robot; Reinforcement; Human–computer interaction; Cognition; Artificial intelligence; Psychology; Computer network; Social psychology; Neuroscience","score_opus":0.036249807818978526,"score_gpt":0.308985041001481,"score_spread":0.2727352331825025,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405938060","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01912405,0.0015919383,0.9729758,0.0004911425,0.0001114819,0.000062092884,0.000016079586,0.00028188343,0.0053455383],"genre_scores_gemma":[0.80660653,0.0012973481,0.18793756,0.00021516417,0.000106565814,0.00022848757,0.00003692466,0.000045825298,0.0035256546],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99953234,0.00020652682,0.000021914528,0.00006318006,0.00013526971,0.00004078394],"domain_scores_gemma":[0.99887544,0.00075200247,0.00010427928,0.000064299646,0.00015642474,0.00004749286],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009047756,0.0005143841,0.000514938,0.0003229509,0.00032242606,0.0007459181,0.00087578857,0.00078773656,0.001404635],"category_scores_gemma":[0.0031884857,0.00027077668,0.00057521777,0.000370827,0.0010689583,0.00083064876,0.00059387763,0.0014639037,0.00025770304],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005627474,0.000080703234,0.00051195343,0.00012377322,0.000040360374,0.00009511287,0.00012441452,0.88261,0.0026468248,0.053580802,0.0006866981,0.05944304],"study_design_scores_gemma":[0.000014894525,0.000063580126,0.00013591172,0.000015839134,0.000008252283,0.00002669536,0.0000127006815,0.97872275,0.000677749,0.0186308,0.0016813575,0.000009483114],"about_ca_topic_score_codex":0.005557662,"about_ca_topic_score_gemma":0.002881618,"teacher_disagreement_score":0.005557662,"about_ca_system_score_codex":0.0010135483,"about_ca_system_score_gemma":0.0013621879,"threshold_uncertainty_score":0.011050642},"labels":[],"label_agreement":null},{"id":"W4405956545","doi":"10.48550/arxiv.2412.20568","title":"Derivations of Animal Movement Models with Explicit Memory","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Movement (music); Computer science; Cognitive science; Neuroscience; Cognitive psychology; Psychology; Philosophy","score_opus":0.08297542405691738,"score_gpt":0.18774141870454697,"score_spread":0.10476599464762959,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405956545","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04147085,0.0009833261,0.9316905,0.00091920083,0.00009076175,0.000039773087,0.00030236354,0.00016932952,0.024333889],"genre_scores_gemma":[0.835984,0.0013725964,0.13590086,0.00042478592,0.00009525036,0.00031739855,0.000381117,0.00015753042,0.02536643],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99987495,0.000035832287,0.000011313043,0.00002428653,0.000036286594,0.000017331642],"domain_scores_gemma":[0.99941707,0.00027335685,0.000099676305,0.00007222101,0.00010515204,0.000032393815],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00050779915,0.00057809503,0.0005104598,0.00058223313,0.00031090766,0.0007217055,0.0012416389,0.0013356495,0.0034282452],"category_scores_gemma":[0.002676942,0.00029648747,0.00087411114,0.00043830372,0.00073174597,0.0011194465,0.0010004725,0.0008730536,0.0008675216],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000010460096,0.000024058461,0.0006694857,0.0000624304,0.000018736177,0.00015923563,0.00009529573,0.70113856,0.001294708,0.28762522,0.0011398707,0.007761939],"study_design_scores_gemma":[0.0000044175513,0.00000671,0.00010269973,0.000009712344,0.000004671888,0.000024727537,0.000008862694,0.95524967,0.00013233394,0.04355931,0.00089139165,0.000005575266],"about_ca_topic_score_codex":0.0058677606,"about_ca_topic_score_gemma":0.0044053528,"teacher_disagreement_score":0.0058677606,"about_ca_system_score_codex":0.00090471184,"about_ca_system_score_gemma":0.00075678324,"threshold_uncertainty_score":0.011667192},"labels":[],"label_agreement":null},{"id":"W4406121228","doi":"10.1007/s11633-023-1482-0","title":"Latent Landmark Graph for Efficient Exploration-exploitation Balance in Hierarchical Reinforcement Learning","year":2025,"lang":"en","type":"article","venue":"Machine Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Landmark; Reinforcement learning; Computer science; Reinforcement; Graph; Artificial intelligence; Balance (ability); Machine learning; Cognitive psychology; Psychology; Theoretical computer science; Social psychology; Neuroscience","score_opus":0.0638528454781579,"score_gpt":0.3792239764301926,"score_spread":0.31537113095203473,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406121228","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028205095,0.00011044426,0.96935576,0.00018526772,0.000023305169,0.000045284498,0.000071868955,0.0004531817,0.0015497839],"genre_scores_gemma":[0.8826362,0.000088127,0.113538906,0.00014082628,0.000027477934,0.00021695219,0.00016686798,0.00014586888,0.0030387673],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99957925,0.0001595155,0.000016653485,0.00008443171,0.00008359094,0.000076505574],"domain_scores_gemma":[0.9977876,0.0015140946,0.00017193671,0.00016630019,0.00019462814,0.00016544944],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010747542,0.0007209634,0.0014287797,0.0006664355,0.00054179877,0.00067829195,0.0017421591,0.001274177,0.0053888517],"category_scores_gemma":[0.0053117466,0.0005579423,0.00037856156,0.00058538147,0.0011589935,0.0016273453,0.0018106564,0.0016654943,0.0005635515],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017808503,0.00010592394,0.0006303976,0.00008303803,0.00003271088,0.000051601746,0.00008427727,0.90537304,0.0023096618,0.042693052,0.0023163976,0.04614186],"study_design_scores_gemma":[0.000012743331,0.000017702727,0.000034263016,0.000002975209,0.0000029773266,0.0000030828542,0.0000031304735,0.9914146,0.0001232001,0.008299986,0.00008262662,0.0000026533774],"about_ca_topic_score_codex":0.0040328265,"about_ca_topic_score_gemma":0.0060291775,"teacher_disagreement_score":0.0053888517,"about_ca_system_score_codex":0.001018843,"about_ca_system_score_gemma":0.0016098766,"threshold_uncertainty_score":0.018027544},"labels":[],"label_agreement":null},{"id":"W4406132533","doi":"10.1007/s00521-024-10829-4","title":"Do as you teach: a multi-teacher approach to self-play in deep reinforcement learning","year":2025,"lang":"en","type":"article","venue":"Neural Computing and Applications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Context (archaeology); Curriculum; Reinforcement; Computational Science and Engineering; Artificial intelligence; State space; Space (punctuation); Baseline (sea); Mathematics education; Machine learning; Psychology; Pedagogy; Mathematics","score_opus":0.019547195795289605,"score_gpt":0.2921890384630919,"score_spread":0.2726418426678023,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406132533","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02650362,0.00013215168,0.96903574,0.0006561167,0.000072882314,0.00006378479,0.00003118421,0.00049827027,0.0030062627],"genre_scores_gemma":[0.8256607,0.00008229955,0.16749631,0.00021492013,0.000042904536,0.00016337169,0.000038526847,0.00014102348,0.0061599207],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994692,0.0002994251,0.000016290602,0.000081647675,0.00007897182,0.000054458676],"domain_scores_gemma":[0.9982948,0.0011091749,0.00011073956,0.00015691425,0.00016267122,0.00016563635],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018962439,0.00046462278,0.0006302363,0.00023674809,0.0004376545,0.00065254,0.0016350556,0.0011636484,0.0041271932],"category_scores_gemma":[0.0047153495,0.00040349519,0.000309672,0.0001881654,0.0009613004,0.001296321,0.0017082512,0.0022878349,0.000391451],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00062822975,0.0006613072,0.003353604,0.00014844156,0.00012904589,0.0001483848,0.00061575894,0.6423069,0.005953283,0.100438885,0.005928086,0.23968802],"study_design_scores_gemma":[0.000014252705,0.000026286261,0.00005707866,0.0000042063393,0.0000056171316,0.000006088495,0.000011796227,0.99071634,0.00042304953,0.008361473,0.00037067255,0.000003174403],"about_ca_topic_score_codex":0.003281498,"about_ca_topic_score_gemma":0.005359975,"teacher_disagreement_score":0.0041271932,"about_ca_system_score_codex":0.0007518722,"about_ca_system_score_gemma":0.001047201,"threshold_uncertainty_score":0.0138068795},"labels":[],"label_agreement":null},{"id":"W4406171015","doi":"10.1109/tase.2025.3527327","title":"Behaviorally-Aware Multi-Agent RL With Dynamic Optimization for Autonomous Driving","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Automation Science and Engineering","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Autonomous agent; Vehicle dynamics; Control engineering; Engineering; Artificial intelligence; Automotive engineering","score_opus":0.011070599583617275,"score_gpt":0.2564603295386846,"score_spread":0.2453897299550673,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406171015","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035633232,0.00025743726,0.9614062,0.00018163276,0.00003584531,0.000030872223,0.000016245122,0.00035769993,0.00208095],"genre_scores_gemma":[0.95945895,0.0000859772,0.03917617,0.000060156657,0.000020307509,0.00006640857,0.000025426229,0.000032559,0.0010739441],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997261,0.0000992581,0.000013080326,0.000052321753,0.000064235566,0.00004493111],"domain_scores_gemma":[0.9995552,0.0002081617,0.00008351865,0.00003561794,0.00008020957,0.000037334543],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00076622266,0.0007270915,0.0006669139,0.00030526306,0.00026475987,0.0005429547,0.000987918,0.0005025831,0.00063612574],"category_scores_gemma":[0.001604032,0.00037005925,0.00042933144,0.00020572965,0.0006289482,0.000520715,0.0009262983,0.0007792109,0.00014739958],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000014759853,0.000020664313,0.00031287505,0.000012735115,0.000017622357,0.000019803847,0.000018363926,0.9881408,0.0007486788,0.0019525538,0.00013793174,0.008603196],"study_design_scores_gemma":[0.0000017824119,0.0000063746493,0.000022487524,6.6320416e-7,0.0000012907632,0.0000015508608,0.0000011828047,0.99936515,0.000055315475,0.0004913003,0.00005199948,9.677653e-7],"about_ca_topic_score_codex":0.0053379266,"about_ca_topic_score_gemma":0.0036737025,"teacher_disagreement_score":0.0053379266,"about_ca_system_score_codex":0.0005999877,"about_ca_system_score_gemma":0.00097288546,"threshold_uncertainty_score":0.01061368},"labels":[],"label_agreement":null},{"id":"W4406209209","doi":"10.2139/ssrn.5024095","title":"Reinforcement Learning in Non-Markov Market-Making","year":2025,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Reinforcement learning; Computer science; Markov decision process; Mathematical optimization; Maximization; Artificial intelligence; Limit (mathematics); Markov process; Mathematics","score_opus":0.008008819207830534,"score_gpt":0.2608087152248799,"score_spread":0.2527998960170494,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406209209","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19843562,0.0010777542,0.7786125,0.0034645079,0.0002509595,0.000109368855,0.00026773583,0.00030407426,0.017477484],"genre_scores_gemma":[0.9647666,0.0005276936,0.022122584,0.00019769008,0.00011606071,0.00013577545,0.00010959031,0.00005282603,0.011971131],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998575,0.00081733684,0.000063346284,0.00020613546,0.00014919785,0.00018902005],"domain_scores_gemma":[0.98229206,0.015432266,0.0007498315,0.0003336526,0.00050099206,0.0006913227],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003480919,0.0008181288,0.0023303009,0.0005862061,0.0008029554,0.0022796558,0.0020763483,0.0026667363,0.008324804],"category_scores_gemma":[0.018220136,0.0008290491,0.00091166905,0.0007184559,0.002831975,0.0036446291,0.0018245537,0.0026474318,0.00048573982],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023234297,0.00013777462,0.0011012603,0.00013614522,0.00009491388,0.00016090044,0.0001243978,0.6501252,0.00046730143,0.3343377,0.0018257839,0.011256178],"study_design_scores_gemma":[0.00004577897,0.000025147008,0.000103217535,0.000007112134,0.000008556231,0.000010678522,0.000013038782,0.8440552,0.0000578755,0.15546142,0.00020229492,0.000009694684],"about_ca_topic_score_codex":0.007416334,"about_ca_topic_score_gemma":0.0052589313,"teacher_disagreement_score":0.008324804,"about_ca_system_score_codex":0.0019435689,"about_ca_system_score_gemma":0.0016378796,"threshold_uncertainty_score":0.027849257},"labels":[],"label_agreement":null},{"id":"W4406267575","doi":"10.1177/02783649241312699","title":"Shared autonomy policy fine-tuning and alignment for robotic tasks","year":2025,"lang":"en","type":"article","venue":"The International Journal of Robotics Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University; McGill University","funders":"","keywords":"Reinforcement learning; Arbitration; Autonomy; Computer science; Human–computer interaction; Task (project management); Robot; Artificial intelligence; Controller (irrigation); Human–robot interaction; Distributed computing; Engineering; Systems engineering","score_opus":0.07345984979332178,"score_gpt":0.40580684606781176,"score_spread":0.33234699627449,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406267575","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017475536,0.000102017686,0.98090196,0.0000814685,0.000019571933,0.00004900831,0.000010092741,0.00036476145,0.0009956069],"genre_scores_gemma":[0.8608999,0.00006894762,0.13774481,0.00009492838,0.000023042288,0.000112854796,0.00002933481,0.00008156729,0.0009445666],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985738,0.0004675641,0.00007716246,0.00036496096,0.00033351657,0.00018304537],"domain_scores_gemma":[0.9979996,0.00090540777,0.00034800664,0.0003480373,0.00024159938,0.0001572552],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022478043,0.00077241636,0.00074375776,0.00031671068,0.00048452706,0.0008368216,0.0013546253,0.00082941446,0.0014793088],"category_scores_gemma":[0.0061045014,0.00032484502,0.00043154534,0.00021223677,0.0012182212,0.0012840721,0.0016452842,0.0015442829,0.00032896257],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016101508,0.00020027568,0.0012659002,0.00007991321,0.00005736418,0.000072012175,0.00025977855,0.8849098,0.008543181,0.014122367,0.00061414856,0.08971425],"study_design_scores_gemma":[0.000018678425,0.0000866814,0.00027091126,0.0000073525675,0.000008657891,0.000025281479,0.000022959262,0.98645914,0.0020391617,0.010342232,0.00070757046,0.000011342768],"about_ca_topic_score_codex":0.0020204547,"about_ca_topic_score_gemma":0.0019464857,"teacher_disagreement_score":0.0022478043,"about_ca_system_score_codex":0.00079403515,"about_ca_system_score_gemma":0.0017123194,"threshold_uncertainty_score":0.011887729},"labels":[],"label_agreement":null},{"id":"W4407253672","doi":"10.1007/978-981-97-8702-9_20","title":"Feature-Based Explainable Reinforcement Learning in Environments with Multiple Sources of Risk","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Reinforcement learning; Feature (linguistics); Artificial intelligence; Reinforcement; Engineering","score_opus":0.00814257528352944,"score_gpt":0.2077444555139946,"score_spread":0.19960188023046516,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407253672","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030482637,0.00019489261,0.96656036,0.00020021881,0.000031322117,0.00001689423,0.00003269424,0.00027372513,0.0022073474],"genre_scores_gemma":[0.8976675,0.00019862341,0.097237945,0.000058401405,0.000041128784,0.000082071274,0.00006342497,0.0000772503,0.0045736],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99974793,0.00007854207,0.000011785418,0.00005461525,0.000064407745,0.000042750664],"domain_scores_gemma":[0.99828255,0.0012905196,0.0001557631,0.00009947262,0.000104603925,0.00006704119],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007064655,0.0006491038,0.0009000268,0.0002527141,0.00024858347,0.0006800992,0.0012468576,0.0011510645,0.0023488821],"category_scores_gemma":[0.003551481,0.000432662,0.0004580336,0.00033890238,0.00077596714,0.0011887397,0.0013424213,0.0016424373,0.00020510584],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000051398627,0.00002283697,0.00025321278,0.000034364937,0.000022737919,0.000056691824,0.00004530392,0.9501279,0.0009171403,0.022642847,0.00049274444,0.025332784],"study_design_scores_gemma":[0.0000045446845,0.00001240929,0.000049844457,0.0000022899255,0.000002570822,0.000007839118,0.0000017296758,0.9880591,0.000108069544,0.011655677,0.000093094095,0.0000028859963],"about_ca_topic_score_codex":0.0023537166,"about_ca_topic_score_gemma":0.002534623,"teacher_disagreement_score":0.0023537166,"about_ca_system_score_codex":0.0007735854,"about_ca_system_score_gemma":0.00048945803,"threshold_uncertainty_score":0.00785774},"labels":[],"label_agreement":null},{"id":"W4407666659","doi":"10.1051/itmconf/20257301007","title":"Optimizing Robotic Arm Learning: Curiosity-Driven Deep Deterministic Policy Gradient","year":2025,"lang":"en","type":"article","venue":"ITM Web of Conferences","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canada Research Chairs","funders":"","keywords":"Curiosity; Artificial intelligence; Computer science; Deep learning; Psychology; Human–computer interaction; Neuroscience","score_opus":0.021392008153920644,"score_gpt":0.2807801220522062,"score_spread":0.25938811389828553,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407666659","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.105328746,0.0005281955,0.88751405,0.00052180723,0.00008472982,0.00008228429,0.000028188608,0.0008133585,0.005098617],"genre_scores_gemma":[0.93294513,0.00012381314,0.06471291,0.00021818739,0.000018614837,0.00009387479,0.00003947716,0.000049200265,0.0017988208],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996958,0.00011448599,0.000013788886,0.00005170036,0.00007026791,0.00005395988],"domain_scores_gemma":[0.9991998,0.0004898115,0.000086439715,0.00005469849,0.00011130688,0.000057919806],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010340721,0.0006336772,0.00058185507,0.0002242551,0.00020090395,0.00042864034,0.0008815406,0.0006715277,0.0011179716],"category_scores_gemma":[0.003263152,0.0003046464,0.00029742828,0.0001705321,0.00066235295,0.00052115245,0.00091897906,0.00096903066,0.0001964786],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000085528234,0.000065010194,0.0010101682,0.00005292711,0.000030512701,0.00005989248,0.000052914642,0.9507404,0.0021646374,0.0039389855,0.00060852966,0.04119044],"study_design_scores_gemma":[0.0000071850695,0.000029732475,0.00005419152,0.0000035383507,0.0000028026877,0.000006906414,0.0000024284288,0.998334,0.00035030427,0.0010358694,0.00017092668,0.0000021573144],"about_ca_topic_score_codex":0.0029599145,"about_ca_topic_score_gemma":0.0024772296,"teacher_disagreement_score":0.0029599145,"about_ca_system_score_codex":0.0005472857,"about_ca_system_score_gemma":0.0011068459,"threshold_uncertainty_score":0.0058853626},"labels":[],"label_agreement":null},{"id":"W4407786710","doi":"10.1007/s12530-025-09660-6","title":"Gradual task complexity scaling (GTCS-DRL): a deep reinforcement learning approach for training automated guided vehicle system","year":2025,"lang":"en","type":"article","venue":"Evolving Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Rimouski","funders":"","keywords":"Reinforcement learning; Computer science; Task (project management); Complex system; Scaling; Training (meteorology); Artificial intelligence; Reinforcement; Systems engineering; Psychology; Physics; Engineering","score_opus":0.05230198632052235,"score_gpt":0.2874851340864007,"score_spread":0.23518314776587834,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407786710","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06964779,0.0005405025,0.92285675,0.00029666393,0.00015113117,0.00013576036,0.00013484847,0.0025561254,0.0036805044],"genre_scores_gemma":[0.8410524,0.000111748224,0.15475203,0.00024037047,0.000032185515,0.00017237516,0.00024196458,0.00016830022,0.0032285752],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998215,0.000044608805,0.000008996675,0.00004708592,0.00004553542,0.000032390053],"domain_scores_gemma":[0.9993832,0.0003123163,0.000052750744,0.000061191466,0.0001310217,0.000059473397],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007141012,0.0008247403,0.0006660104,0.00032654783,0.0002504741,0.00038552113,0.0012962865,0.0009233279,0.0024869991],"category_scores_gemma":[0.0017250263,0.00044476386,0.0004113333,0.00025082112,0.0004907231,0.00049090135,0.0010154886,0.0020088914,0.00038704788],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000090344,0.00012261815,0.00079061545,0.00007886905,0.00003951628,0.00005063332,0.000053746902,0.86133564,0.0039380714,0.0024591994,0.0025386994,0.12850201],"study_design_scores_gemma":[0.0000054291245,0.000024011142,0.00004530866,0.0000026785995,0.0000022512552,0.0000030516119,0.0000016098771,0.9991116,0.00030128515,0.0003302605,0.00017092252,0.0000016406009],"about_ca_topic_score_codex":0.01106549,"about_ca_topic_score_gemma":0.015601081,"teacher_disagreement_score":0.01106549,"about_ca_system_score_codex":0.0006874763,"about_ca_system_score_gemma":0.0011284084,"threshold_uncertainty_score":0.02200216},"labels":[],"label_agreement":null},{"id":"W4407857640","doi":"10.1145/3696443.3708963","title":"Vectron: A Dynamic Programming Auto-vectorization Framework","year":2025,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Vectorization (mathematics); Computer science; Dynamic programming; Programming language; Parallel computing; Algorithm","score_opus":0.005537035072222719,"score_gpt":0.268197002878428,"score_spread":0.2626599678062053,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407857640","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0014973024,0.00010164842,0.9911343,0.00007371386,0.000040500483,0.000040640894,0.00013522594,0.004642914,0.0023337167],"genre_scores_gemma":[0.07880236,0.00035014033,0.9116553,0.00025602744,0.00006736255,0.00050962705,0.0011195748,0.0022002256,0.0050394055],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99932075,0.00016726168,0.000044400094,0.00014634151,0.00023916623,0.00008207731],"domain_scores_gemma":[0.99933463,0.00028733426,0.00005395315,0.00010567645,0.00017157136,0.000046797275],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008532522,0.0010752717,0.00079471525,0.00058408873,0.0004163772,0.0013075486,0.0025643443,0.00080294925,0.008909324],"category_scores_gemma":[0.0023036692,0.0006432076,0.0009928122,0.0007631901,0.0007974914,0.0014140542,0.0016068093,0.0023915228,0.0026738949],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014994972,0.00013709861,0.00085016247,0.00030823957,0.00010605239,0.00011784784,0.00010914829,0.57602865,0.007747158,0.09006474,0.023069955,0.30131105],"study_design_scores_gemma":[0.000026367721,0.00003175481,0.000059594455,0.000015814756,0.000007879386,0.000030211806,0.00001217036,0.9690986,0.0017248013,0.018420106,0.010560697,0.000011901979],"about_ca_topic_score_codex":0.0048643234,"about_ca_topic_score_gemma":0.007893519,"teacher_disagreement_score":0.008909324,"about_ca_system_score_codex":0.00072904985,"about_ca_system_score_gemma":0.0022164762,"threshold_uncertainty_score":0.029804647},"labels":[],"label_agreement":null},{"id":"W4407870583","doi":"10.3390/risks13030040","title":"Deep Reinforcement Learning in Non-Markov Market-Making","year":2025,"lang":"en","type":"article","venue":"Risks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs","keywords":"Reinforcement learning; Artificial intelligence; Markov chain; Reinforcement; Computer science; Economics; Machine learning; Psychology; Social psychology","score_opus":0.016540782736329213,"score_gpt":0.2945006458921195,"score_spread":0.2779598631557903,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407870583","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05419444,0.00029552597,0.9394695,0.0009073424,0.0000672417,0.00003410573,0.00006353131,0.0003288051,0.0046395473],"genre_scores_gemma":[0.915987,0.00020262721,0.07947106,0.00020146456,0.000041820032,0.00008275969,0.000061698935,0.000060161205,0.0038913],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99958676,0.00016929755,0.000019546602,0.0000758126,0.00007517873,0.00007343799],"domain_scores_gemma":[0.9982835,0.0011763332,0.00018383843,0.00009477678,0.00015037144,0.000111078996],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015175606,0.0005852384,0.0009276567,0.0002980204,0.00033116402,0.0010460892,0.0011713315,0.0011415739,0.002936218],"category_scores_gemma":[0.0050983694,0.00041198652,0.00050886394,0.00029548656,0.0013650898,0.0014491769,0.0010782321,0.0018735566,0.00027808078],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000016834558,0.000019393561,0.0003175025,0.000022252581,0.000014368277,0.000032071537,0.000029026462,0.96594477,0.00036170604,0.027892923,0.00027632594,0.0050729183],"study_design_scores_gemma":[0.000003597528,0.000005434945,0.00002775259,0.0000022584536,0.0000013184251,0.000002733897,0.0000017505619,0.9904288,0.00008955836,0.00933496,0.00010006004,0.0000017002857],"about_ca_topic_score_codex":0.0050927466,"about_ca_topic_score_gemma":0.0048818905,"teacher_disagreement_score":0.0050927466,"about_ca_system_score_codex":0.0013526401,"about_ca_system_score_gemma":0.001267784,"threshold_uncertainty_score":0.0101261735},"labels":[],"label_agreement":null},{"id":"W4407914197","doi":"10.1007/978-3-031-81596-6_8","title":"Memory Augmented Multi-agent Reinforcement Learning for Cooperative Environment","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Computer science; Reinforcement learning; Human–computer interaction; Artificial intelligence","score_opus":0.022046313600716162,"score_gpt":0.25488716534788347,"score_spread":0.23284085174716732,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407914197","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030228226,0.00038817976,0.964469,0.000086171756,0.000069153844,0.000030120571,0.000019057721,0.0004608478,0.004249199],"genre_scores_gemma":[0.9248641,0.00016547846,0.06923698,0.000046153513,0.000031998796,0.000103667204,0.000040548526,0.000043258548,0.0054678195],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99981123,0.000053416396,0.0000096805015,0.00004254943,0.00004671928,0.000036392197],"domain_scores_gemma":[0.9995604,0.00023759993,0.000045889064,0.000051354124,0.000074611286,0.0000300982],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00050170074,0.00056989107,0.0007622641,0.00021369838,0.00032290502,0.0005599767,0.0012497107,0.0007451803,0.002534224],"category_scores_gemma":[0.0012236072,0.00026350172,0.0003397893,0.00023297252,0.00053852453,0.00070094725,0.0014417794,0.0010000422,0.00037322342],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010738824,0.00007997322,0.00019588294,0.00006530375,0.000033119777,0.00007365694,0.00006591977,0.9069747,0.0034574852,0.0076329517,0.0010796076,0.080233924],"study_design_scores_gemma":[0.0000064677934,0.000029015071,0.00003506701,0.0000022379545,0.0000034061948,0.000008418073,0.0000031889301,0.99707806,0.00035868763,0.0022334848,0.0002395157,0.0000024440515],"about_ca_topic_score_codex":0.002894552,"about_ca_topic_score_gemma":0.0029304395,"teacher_disagreement_score":0.002894552,"about_ca_system_score_codex":0.00043387298,"about_ca_system_score_gemma":0.00043487397,"threshold_uncertainty_score":0.008477807},"labels":[],"label_agreement":null},{"id":"W4407951213","doi":"10.1109/cdc56724.2024.10886523","title":"Design of Denial-of-Service Attack Strategy With Energy Constraint: An Approximate Dynamic Programming Approach","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"National Natural Science Foundation of China","keywords":"Computer science; Denial-of-service attack; Constraint (computer-aided design); Denial; Dynamic programming; Computer security; Service (business); Mathematical optimization; Distributed computing; Engineering; Algorithm; Operating system; Mathematics; Business; The Internet","score_opus":0.03830944495505354,"score_gpt":0.2737495941951317,"score_spread":0.23544014924007814,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407951213","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006071369,0.00019522425,0.99089336,0.00016278899,0.000018078714,0.00004204185,0.000014439902,0.00008117988,0.0025215528],"genre_scores_gemma":[0.83185023,0.0005122351,0.162897,0.00022467473,0.000045263063,0.00041739867,0.00006229033,0.00008413587,0.0039067306],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994999,0.00014547656,0.000021703361,0.00010346593,0.00013684125,0.000092615795],"domain_scores_gemma":[0.99924856,0.00044649746,0.000098248696,0.000027948692,0.00013541586,0.000043254768],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010083738,0.001025741,0.0014303665,0.00052323606,0.0003444704,0.0011566464,0.0009453097,0.0012212587,0.0020163495],"category_scores_gemma":[0.0022108362,0.0005708245,0.0005392087,0.00046683004,0.00083473825,0.0008329473,0.00094466173,0.0011474666,0.0002713729],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000022373613,0.000014927554,0.00014044266,0.000050682265,0.00001373403,0.000027431684,0.000024005469,0.9806836,0.00082347426,0.0071760314,0.00027957084,0.010743669],"study_design_scores_gemma":[0.0000037418295,0.0000137808875,0.000016962118,0.0000034123004,0.0000025939096,0.000006594094,0.000003605493,0.99836916,0.000098654935,0.0013271,0.00015265046,0.0000017228598],"about_ca_topic_score_codex":0.004179421,"about_ca_topic_score_gemma":0.0021837375,"teacher_disagreement_score":0.004179421,"about_ca_system_score_codex":0.00094755855,"about_ca_system_score_gemma":0.0017064046,"threshold_uncertainty_score":0.008310199},"labels":[],"label_agreement":null},{"id":"W4408012843","doi":"10.1007/978-981-96-0077-9_5","title":"Evolving Many-Model Agents with Vector and Matrix Operations in Tangled Program Graphs","year":2025,"lang":"en","type":"book-chapter","venue":"Genetic and evolutionary computation","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Matrix (chemical analysis); Vector (molecular biology); Computer science; Mathematics; Biology; Materials science; Genetics","score_opus":0.013447265442516666,"score_gpt":0.2528431059099153,"score_spread":0.23939584046739865,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408012843","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20060083,0.00028173282,0.7825656,0.00037725767,0.00008924615,0.000060662813,0.000078384044,0.00069433026,0.015251949],"genre_scores_gemma":[0.77866954,0.00019741777,0.21214165,0.000088979425,0.00002228552,0.00010434514,0.00010472398,0.00024596843,0.008425083],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99986255,0.000060036236,0.0000060372663,0.000026105465,0.000029628543,0.000015478987],"domain_scores_gemma":[0.9993917,0.00041321217,0.000048246504,0.00005014609,0.000045165387,0.000051610263],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003974752,0.00046339844,0.00059523486,0.00048088958,0.00052734895,0.000692679,0.0010170019,0.0008763398,0.0030638697],"category_scores_gemma":[0.001878876,0.00047356554,0.0005210879,0.00049212185,0.0010301743,0.0014794095,0.0010867096,0.0010132992,0.00022711654],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003003242,0.000029767263,0.00017315014,0.000030127887,0.000014086226,0.000059217455,0.00007119139,0.91689557,0.0011684481,0.06623291,0.0005973987,0.014698069],"study_design_scores_gemma":[0.0000060536986,0.000010760929,0.000019526804,0.000003057805,0.0000021973417,0.0000075162657,0.000009431171,0.9673411,0.0001888212,0.0321462,0.00026280087,0.0000025143827],"about_ca_topic_score_codex":0.0028004874,"about_ca_topic_score_gemma":0.0033819359,"teacher_disagreement_score":0.0030638697,"about_ca_system_score_codex":0.00076857634,"about_ca_system_score_gemma":0.00041538727,"threshold_uncertainty_score":0.010249615},"labels":[],"label_agreement":null},{"id":"W4408261083","doi":"10.1515/9781772127867","title":"Digital Memory Agents in Canada","year":2024,"lang":"en","type":"book","venue":"University of Alberta Press eBooks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science","score_opus":0.01799329881715726,"score_gpt":0.18708337127741648,"score_spread":0.16909007246025923,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408261083","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025714837,0.014466808,0.00069604674,0.00993343,0.0007181506,0.00004034109,0.00026105938,0.00014934,0.94802],"genre_scores_gemma":[0.27234244,0.014621979,0.0009961639,0.0012297211,0.000103808794,0.000030372315,0.0002790369,0.0001230453,0.7102734],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99926454,0.0000491252,0.000015236184,0.000059446826,0.00031170674,0.00029991553],"domain_scores_gemma":[0.9992986,0.00007934111,0.00002957407,0.00004770326,0.0003049137,0.00023983949],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00034014843,0.0005116506,0.00028314302,0.0021046626,0.017638663,0.014206998,0.0014041469,0.0017575268,0.023798687],"category_scores_gemma":[0.0011339999,0.00027113265,0.00025770746,0.004288332,0.007520223,0.0025363537,0.0032950523,0.0016152128,0.0023556624],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005424839,0.000035335735,0.0015271577,0.00022733508,0.000008847854,0.00078060676,0.04663089,0.00050293864,0.00041983463,0.6154505,0.23597746,0.098384865],"study_design_scores_gemma":[0.0000018774298,0.000002985847,0.0004149512,0.00006540577,0.0000039219817,0.00007254806,0.00675378,0.00007656014,0.000108115055,0.0017189715,0.9907733,0.000007675075],"about_ca_topic_score_codex":0.976995,"about_ca_topic_score_gemma":0.9876172,"teacher_disagreement_score":0.09538903,"about_ca_system_score_codex":0.09538903,"about_ca_system_score_gemma":0.08294031,"threshold_uncertainty_score":0.6920991},"labels":[],"label_agreement":null},{"id":"W4408363681","doi":"10.1007/s10846-025-02235-2","title":"Risk-Sensitive Autonomous Exploration of Unknown Environments: A Deep Reinforcement Learning Perspective","year":2025,"lang":"en","type":"article","venue":"Journal of Intelligent & Robotic Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada; Government of Alberta","keywords":"Reinforcement learning; Perspective (graphical); Computer science; Artificial intelligence; Reinforcement; Human–computer interaction; Psychology; Social psychology","score_opus":0.018578048339738428,"score_gpt":0.2630788909470452,"score_spread":0.24450084260730676,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408363681","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06481378,0.0005633626,0.9305163,0.0003984378,0.00003876801,0.000039494505,0.0000291016,0.00020877953,0.0033919967],"genre_scores_gemma":[0.9570252,0.00026139154,0.041058555,0.00007826922,0.000020905887,0.000057634887,0.000025279824,0.000029643968,0.0014431779],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99967825,0.00011697757,0.000013921913,0.00006055997,0.00007364593,0.000056679466],"domain_scores_gemma":[0.99883264,0.0007118269,0.00016563002,0.00007906246,0.00012821642,0.000082595136],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009688146,0.000616496,0.0006764518,0.00025153358,0.00022516458,0.00068151136,0.00077204074,0.00064791087,0.00083792914],"category_scores_gemma":[0.0031289419,0.00029391903,0.0003387189,0.00018423267,0.00079693453,0.0008798368,0.0009883835,0.0010410001,0.00011650356],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000036593527,0.000024572466,0.0006714613,0.000035342902,0.000020149957,0.00003935569,0.000036250767,0.97886056,0.0012347132,0.006230304,0.00019793106,0.012612722],"study_design_scores_gemma":[0.0000033395308,0.000018954914,0.00007167112,0.0000038483977,0.000002994884,0.000007743936,0.0000047244644,0.9972192,0.00022106638,0.0023049074,0.00013911961,0.0000024298174],"about_ca_topic_score_codex":0.0031305961,"about_ca_topic_score_gemma":0.0017136446,"teacher_disagreement_score":0.0031305961,"about_ca_system_score_codex":0.0005655214,"about_ca_system_score_gemma":0.0010848225,"threshold_uncertainty_score":0.0062247515},"labels":[],"label_agreement":null},{"id":"W4408434866","doi":"10.1016/j.eswa.2025.127180","title":"Application of Soft Actor-Critic algorithms in optimizing wastewater treatment with time delays integration","year":2025,"lang":"en","type":"article","venue":"Expert Systems with Applications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kruger (Canada)","funders":"H2020 Marie Skłodowska-Curie Actions; Horizon 2020; Research Executive Agency; Horizon 2020 Framework Programme; European Commission","keywords":"Computer science; Algorithm; Mathematical optimization; Artificial intelligence; Mathematics","score_opus":0.009996203683530259,"score_gpt":0.25450466053906284,"score_spread":0.24450845685553257,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408434866","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.074684076,0.00048975745,0.91834843,0.0004302291,0.00010404939,0.000047268324,0.000047434005,0.00075330446,0.0050955047],"genre_scores_gemma":[0.9531183,0.00013233416,0.044291776,0.000121314886,0.000023440218,0.00006488404,0.00005167618,0.000047909656,0.0021483582],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997775,0.000071297436,0.000012309878,0.000053342716,0.000051213105,0.00003435768],"domain_scores_gemma":[0.99925786,0.00046257762,0.00007592982,0.00003907883,0.0001224545,0.000042106945],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006930029,0.0009528593,0.00066968834,0.00026636102,0.00023273875,0.00063301984,0.00071648153,0.0008564501,0.0009938296],"category_scores_gemma":[0.0018938495,0.0004057745,0.00045070055,0.00019820518,0.00061978726,0.00046689544,0.00068441196,0.0010495626,0.00015363314],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000016561788,0.000007916788,0.0001619767,0.000014184188,0.000010671913,0.000018388731,0.0000075812654,0.99411976,0.00053192384,0.0007315003,0.000091898604,0.004287513],"study_design_scores_gemma":[0.0000020181458,0.0000051511374,0.000014763881,0.0000010159765,0.0000013689764,0.0000013790095,7.492728e-7,0.9995419,0.00014604232,0.00023935126,0.000045331246,9.104133e-7],"about_ca_topic_score_codex":0.0075641936,"about_ca_topic_score_gemma":0.0058844592,"teacher_disagreement_score":0.0075641936,"about_ca_system_score_codex":0.00073475734,"about_ca_system_score_gemma":0.0011468716,"threshold_uncertainty_score":0.015040338},"labels":[],"label_agreement":null},{"id":"W4408518028","doi":"10.2139/ssrn.5182425","title":"Sage: Self-Evolving Agents with Reflective and Memory-Augmented Abilities","year":2025,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Psychology; Cognitive psychology; SAGE; Computer science; Cognitive science; Physics","score_opus":0.009187574990174665,"score_gpt":0.2573514460157894,"score_spread":0.24816387102561474,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408518028","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09539163,0.00023802246,0.8797206,0.00040871045,0.0002371549,0.00013252559,0.00020468474,0.0077777635,0.015888777],"genre_scores_gemma":[0.7161959,0.00021503409,0.26327983,0.00018108307,0.000047408037,0.00031824532,0.00035453384,0.00043918754,0.018968962],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99986446,0.00003692216,0.000009128196,0.000027269207,0.00004735951,0.0000148220515],"domain_scores_gemma":[0.99949145,0.00023728647,0.00003887614,0.00009485735,0.000059089274,0.000078403544],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00036258888,0.00035625073,0.00038435528,0.0002136226,0.0002399887,0.00061317376,0.000876013,0.00074208126,0.004760497],"category_scores_gemma":[0.001972366,0.00019432504,0.00032455343,0.0001600242,0.0006644951,0.0007435668,0.0014735803,0.00084479613,0.0011149395],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051437906,0.00030213202,0.0027508435,0.00032792988,0.00013643291,0.00057225354,0.0004616848,0.6335627,0.038489703,0.13133898,0.011651099,0.17989187],"study_design_scores_gemma":[0.00013349552,0.00016661194,0.0002939032,0.000019401348,0.000028108525,0.00015933649,0.0000413747,0.9243333,0.008887995,0.054819703,0.0110989185,0.000017858792],"about_ca_topic_score_codex":0.0004693866,"about_ca_topic_score_gemma":0.00071399607,"teacher_disagreement_score":0.004760497,"about_ca_system_score_codex":0.00017694579,"about_ca_system_score_gemma":0.00040898361,"threshold_uncertainty_score":0.015925467},"labels":[],"label_agreement":null},{"id":"W4408794490","doi":"10.1109/swc62898.2024.00346","title":"Simulating a Multi-Agent UAV System Coordinated by State Machines Using Godot","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary; Athabasca University","funders":"","keywords":"Computer science; State (computer science); Multi-agent system; Distributed computing; Embedded system; Artificial intelligence; Programming language","score_opus":0.030465042359474837,"score_gpt":0.2919060881060475,"score_spread":0.2614410457465727,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408794490","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.39996868,0.00014519422,0.57469463,0.0002741799,0.00013931589,0.00024548292,0.0005470699,0.0024610118,0.021524416],"genre_scores_gemma":[0.9008902,0.00009284401,0.09547724,0.000048918966,0.0000072487846,0.00017150855,0.000241715,0.000098127086,0.0029722909],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998969,0.00002979962,0.000006328485,0.00001528812,0.000033136163,0.00001850745],"domain_scores_gemma":[0.9996153,0.00024713052,0.000029506833,0.000030588642,0.000044154975,0.000033195207],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00029360285,0.00039164675,0.00033109015,0.00024306479,0.0003374207,0.00046354753,0.0006422523,0.0005360837,0.0028390207],"category_scores_gemma":[0.0008408641,0.00021039101,0.0005177665,0.00012574765,0.00051124673,0.0004433019,0.00060842955,0.0004076726,0.00015339613],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000048145543,0.00002310088,0.00078349444,0.000039741273,0.00001503352,0.000069737536,0.00007069131,0.9885649,0.002298484,0.004471947,0.00027152224,0.0033431458],"study_design_scores_gemma":[0.000010666996,0.000021726932,0.00011674299,0.0000027629228,0.0000028426678,0.000008145122,0.0000113427395,0.99773824,0.000856473,0.00062628346,0.00060197845,0.0000027241738],"about_ca_topic_score_codex":0.009129545,"about_ca_topic_score_gemma":0.007114403,"teacher_disagreement_score":0.009129545,"about_ca_system_score_codex":0.00046209386,"about_ca_system_score_gemma":0.0006095835,"threshold_uncertainty_score":0.018152833},"labels":[],"label_agreement":null},{"id":"W4408891814","doi":"10.21203/rs.3.rs-6051145/v1","title":"Inequity Aversion Toward AI Counterparts","year":2025,"lang":"en","type":"preprint","venue":"Research Square","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; University of Toronto","funders":"","keywords":"Inequity aversion; Economics; Psychology; Cognitive psychology; Mathematics; Inequality","score_opus":0.09820482988377287,"score_gpt":0.4300601923562331,"score_spread":0.3318553624724602,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408891814","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5766066,0.00046099603,0.15685561,0.0062045753,0.00020885533,0.00006680439,0.0001253218,0.00026097396,0.25921026],"genre_scores_gemma":[0.9826325,0.000113461116,0.004650105,0.0002954132,0.000035910183,0.000026308231,0.00003062818,0.000025228434,0.012190445],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9989949,0.00044186763,0.000030416084,0.00014376045,0.00025788098,0.00013123112],"domain_scores_gemma":[0.99327785,0.0039340705,0.00073072605,0.0008131969,0.0007529335,0.00049124006],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001683756,0.0003250442,0.0002968532,0.00032384225,0.00048944616,0.0013470651,0.0003386495,0.00064533274,0.013344397],"category_scores_gemma":[0.013590585,0.00012658075,0.00019115533,0.00033881934,0.0010902674,0.0013170105,0.0012300683,0.0020157201,0.0006304537],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058146945,0.000264038,0.010610196,0.00016708567,0.00006442323,0.00015639765,0.0006216867,0.020073477,0.007162886,0.849577,0.0070845624,0.10363677],"study_design_scores_gemma":[0.00008689751,0.00021234203,0.009781791,0.00004241287,0.00004330973,0.0002553641,0.00054044573,0.11344493,0.0036518725,0.8542362,0.017682057,0.000022449945],"about_ca_topic_score_codex":0.0010334286,"about_ca_topic_score_gemma":0.0009757153,"teacher_disagreement_score":0.013344397,"about_ca_system_score_codex":0.0006872215,"about_ca_system_score_gemma":0.0004664235,"threshold_uncertainty_score":0.044641495},"labels":[],"label_agreement":null},{"id":"W4408954824","doi":"10.1007/s00521-025-11100-0","title":"Advances and applications in inverse reinforcement learning: a comprehensive review","year":2025,"lang":"en","type":"review","venue":"Neural Computing and Applications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"RUDN University; Monash University","keywords":"Computational Science and Engineering; Computer science; Reinforcement learning; Reinforcement; Inverse; Artificial intelligence; Machine learning; Applied mathematics; Mathematics; Geometry; Engineering; Structural engineering","score_opus":0.03519127845833417,"score_gpt":0.34493019071533093,"score_spread":0.3097389122569968,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408954824","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000117673575,0.9978708,0.00060497626,0.00023697336,0.000121785815,0.0000070306214,0.000025435036,0.0000119846545,0.0010032662],"genre_scores_gemma":[0.001071463,0.997849,0.0005000834,0.00014510695,0.00011624077,0.00001109422,0.000033483186,0.000003495402,0.00027008675],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99971026,0.000054995304,0.000047704212,0.00006360976,0.000100945515,0.000022500202],"domain_scores_gemma":[0.99861956,0.000937015,0.0001231539,0.00003284992,0.00023957256,0.00004782845],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011322248,0.0010176721,0.0015336842,0.0030050057,0.00027153775,0.0011592314,0.0010330161,0.001131268,0.0041877166],"category_scores_gemma":[0.0024895333,0.0004732348,0.00092499657,0.003452483,0.0004757223,0.0015855763,0.00075461244,0.0014529445,0.0018599975],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005642695,0.00008597651,0.00023235963,0.040776752,0.00016570669,0.00011554849,0.00007571441,0.0022999218,0.0007633528,0.0101488875,0.02401638,0.921263],"study_design_scores_gemma":[0.000021137146,0.00017753753,0.0009413637,0.01602788,0.00032438524,0.0005970213,0.00007466946,0.0008872395,0.00058293977,0.0068279,0.97348356,0.00005431547],"about_ca_topic_score_codex":0.0017964457,"about_ca_topic_score_gemma":0.0022572812,"teacher_disagreement_score":0.0041877166,"about_ca_system_score_codex":0.00072117575,"about_ca_system_score_gemma":0.0017608537,"threshold_uncertainty_score":0.014009237},"labels":[],"label_agreement":null},{"id":"W4409147637","doi":"10.1038/s41586-025-08744-2","title":"Mastering diverse control tasks through world models","year":2025,"lang":"en","type":"article","venue":"Nature","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":82,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Scratch; Reinforcement learning; Artificial intelligence; Robustness (evolution); Normalization (sociology); Machine learning; Control (management); Programming language","score_opus":0.015993840515038285,"score_gpt":0.27037089696356514,"score_spread":0.2543770564485269,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409147637","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.094080746,0.00026261053,0.89723694,0.00043235716,0.000037118545,0.0001228217,0.000072194765,0.0014865958,0.006268593],"genre_scores_gemma":[0.83218086,0.00014792221,0.16414174,0.00016365408,0.000018956662,0.00015456263,0.00014384197,0.0001147946,0.0029336975],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994779,0.00014906185,0.000026239011,0.00018142132,0.00009735746,0.00006800847],"domain_scores_gemma":[0.9985917,0.0006595791,0.00016987635,0.0003428224,0.00012611106,0.00010987294],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014672944,0.0010763145,0.0010089567,0.00037456487,0.00038667687,0.0011970345,0.0015256534,0.0011633193,0.0027533076],"category_scores_gemma":[0.0032486117,0.00056182133,0.00064403814,0.00028028124,0.0015553213,0.001945854,0.0019315409,0.0018071135,0.00045766152],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005182831,0.00006454044,0.00051123835,0.00004309875,0.000036356945,0.000030362304,0.000055355304,0.9532869,0.0020477246,0.0070311776,0.0006844724,0.036156975],"study_design_scores_gemma":[0.000011143531,0.00002624036,0.000062831896,0.0000049621067,0.000003951499,0.000006555206,0.0000075199755,0.99302,0.00059378764,0.005912815,0.00034669877,0.0000034520554],"about_ca_topic_score_codex":0.0025657378,"about_ca_topic_score_gemma":0.0026995402,"teacher_disagreement_score":0.0027533076,"about_ca_system_score_codex":0.0008958358,"about_ca_system_score_gemma":0.00091185624,"threshold_uncertainty_score":0.009210765},"labels":[],"label_agreement":null},{"id":"W4409183409","doi":"10.1007/978-3-031-88714-7_35","title":"Retrieval-Augmented Neural Team Formation","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; University of Guelph","funders":"","keywords":"Computer science; Artificial neural network; Artificial intelligence; Human–computer interaction","score_opus":0.014930001063955195,"score_gpt":0.2470052268019763,"score_spread":0.2320752257380211,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409183409","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021001022,0.00043632975,0.9610148,0.00022179532,0.00020475364,0.00004801287,0.00005810594,0.00077019824,0.016244946],"genre_scores_gemma":[0.7621596,0.00040700345,0.19440605,0.00015145414,0.00015266152,0.00018038016,0.00028825342,0.00022280605,0.04203185],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99979883,0.00004807386,0.000008907564,0.000056926514,0.0000509678,0.000036221427],"domain_scores_gemma":[0.99966145,0.00013232262,0.000038167,0.00007241497,0.00006459495,0.000031047017],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00041207782,0.0005406693,0.0008505321,0.0003098037,0.00039171113,0.0006435951,0.0014173776,0.0011072867,0.0061231614],"category_scores_gemma":[0.0012414248,0.0004503922,0.00045137602,0.00042109916,0.0006677773,0.0010109498,0.0020735955,0.0010230313,0.0013899463],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013912743,0.00007283221,0.00016354944,0.00010045258,0.000043295106,0.00008076203,0.00008252624,0.76713556,0.0077323476,0.031021586,0.0066051884,0.18682283],"study_design_scores_gemma":[0.000009392461,0.000038064984,0.00004499246,0.0000050132844,0.0000046388163,0.00002373893,0.000008892783,0.98722154,0.000973713,0.010756271,0.0009087401,0.000004956679],"about_ca_topic_score_codex":0.0020287896,"about_ca_topic_score_gemma":0.002081531,"teacher_disagreement_score":0.0061231614,"about_ca_system_score_codex":0.0004494482,"about_ca_system_score_gemma":0.00048291465,"threshold_uncertainty_score":0.02048397},"labels":[],"label_agreement":null},{"id":"W4409347970","doi":"10.1609/aaai.v39i22.34490","title":"Efficient Communication in Multi-Agent Reinforcement Learning with Implicit Consensus Generation","year":2025,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Chinese Academy of Sciences","keywords":"Reinforcement learning; Reinforcement; Computer science; Artificial intelligence; Psychology; Social psychology","score_opus":0.08210690364603603,"score_gpt":0.31301082425299065,"score_spread":0.23090392060695464,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409347970","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05137446,0.00012246105,0.9463834,0.00019851548,0.000027644634,0.00006197063,0.000012662382,0.00040445093,0.0014145361],"genre_scores_gemma":[0.93145186,0.00004202844,0.06709504,0.00008223066,0.000023470964,0.00012657416,0.000031466054,0.000034657747,0.0011127327],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99853814,0.00058961683,0.00007676424,0.0002444542,0.0003847455,0.00016629488],"domain_scores_gemma":[0.9954965,0.0026358615,0.0006254495,0.0004804256,0.0005382965,0.00022345968],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002461618,0.00082421425,0.0010810673,0.00043162433,0.00072526105,0.0006911317,0.002214879,0.001103066,0.00088303076],"category_scores_gemma":[0.0077287885,0.00040539232,0.00031187813,0.00041328598,0.0011822607,0.0015152942,0.0022196737,0.0014815087,0.000199261],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017993376,0.00010261414,0.0006485946,0.000054164793,0.000024575238,0.00009752684,0.00018117802,0.93081385,0.002364674,0.008360893,0.00049721904,0.056674723],"study_design_scores_gemma":[0.000022166294,0.000030174288,0.000039751092,0.0000020052667,0.0000026906719,0.000008850157,0.000006231206,0.99680686,0.00050978636,0.0024421022,0.00012621719,0.0000032349492],"about_ca_topic_score_codex":0.0026011383,"about_ca_topic_score_gemma":0.002015983,"teacher_disagreement_score":0.0026011383,"about_ca_system_score_codex":0.0007447994,"about_ca_system_score_gemma":0.0013302591,"threshold_uncertainty_score":0.013018429},"labels":[],"label_agreement":null},{"id":"W4409361576","doi":"10.1609/aaai.v39i25.34864","title":"ModelDiff: Symbolic Dynamic Programming for Model-Aware Policy Transfer in Deep Q-Learning","year":2025,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Transfer of learning; Transfer (computing); Dynamic programming; Artificial intelligence; Programming language; Parallel computing; Algorithm","score_opus":0.04088646043850981,"score_gpt":0.31499908200665955,"score_spread":0.27411262156814975,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409361576","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0039194,0.0001670061,0.9931344,0.00026436843,0.000037579543,0.000043016596,0.000049013852,0.0006564404,0.0017287118],"genre_scores_gemma":[0.5557053,0.0003467247,0.43744537,0.00063438714,0.00008505116,0.00053957733,0.0003288749,0.00047386994,0.0044407854],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990834,0.00040359778,0.000037919872,0.00017072253,0.00020788396,0.00009657807],"domain_scores_gemma":[0.9971999,0.0020974618,0.0001545725,0.00021728851,0.00019257562,0.00013827605],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025187107,0.0010564835,0.0011799172,0.00046990794,0.00043120896,0.0011426797,0.0021614474,0.0016008271,0.0056743883],"category_scores_gemma":[0.008564932,0.0007339686,0.0006888112,0.00056004047,0.0015085168,0.0016072118,0.0028422326,0.0034195744,0.00072670943],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006206595,0.00006071868,0.00036563983,0.00008771971,0.000025554977,0.000047087502,0.00006400926,0.90571254,0.00048590588,0.040265586,0.0022582342,0.050564904],"study_design_scores_gemma":[0.000007958289,0.00000891213,0.000008847668,0.0000041361586,0.0000013657892,0.0000029059106,0.0000021193273,0.9878902,0.00010001165,0.011682794,0.00028926088,0.000001549884],"about_ca_topic_score_codex":0.0055527836,"about_ca_topic_score_gemma":0.006738093,"teacher_disagreement_score":0.0056743883,"about_ca_system_score_codex":0.0015285749,"about_ca_system_score_gemma":0.0034206165,"threshold_uncertainty_score":0.018982708},"labels":[],"label_agreement":null},{"id":"W4409361642","doi":"10.1609/aaai.v39i25.34887","title":"Flow Factorization for Efficient Generative Flow Networks","year":2025,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"University of Toronto","keywords":"Flow (mathematics); Computer science; Generative grammar; Artificial intelligence; Mathematics; Geometry","score_opus":0.05156236654893604,"score_gpt":0.29477046231317067,"score_spread":0.24320809576423463,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409361642","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00410635,0.000098757475,0.9941462,0.00008746641,0.000020729753,0.000045573568,0.000070678136,0.0004632373,0.0009609729],"genre_scores_gemma":[0.3714318,0.00039901453,0.6179561,0.00042789563,0.00009755935,0.0006956468,0.0012248476,0.0004958396,0.0072713504],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99927956,0.00021640188,0.000035923524,0.0002006757,0.0001721516,0.000095290634],"domain_scores_gemma":[0.9976428,0.0015876004,0.00015203489,0.00018525074,0.0003083468,0.0001239419],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00186261,0.0013206953,0.0011560159,0.0012202761,0.000707523,0.0010036425,0.0016901675,0.0015605489,0.0062490953],"category_scores_gemma":[0.0077864802,0.0008292664,0.0010704313,0.0009154938,0.0013079896,0.0018965716,0.0021616307,0.002577879,0.0013739811],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006929553,0.00005380697,0.00084144674,0.0000676639,0.000021378903,0.00006533266,0.00007778826,0.87708956,0.0012866843,0.035347715,0.0023045726,0.08277476],"study_design_scores_gemma":[0.00000501905,0.0000067542373,0.000025128134,0.000004729422,0.0000020930638,0.000008245129,0.0000030319493,0.9882967,0.00022551721,0.010998088,0.00042208715,0.0000026428895],"about_ca_topic_score_codex":0.007560636,"about_ca_topic_score_gemma":0.010759827,"teacher_disagreement_score":0.007560636,"about_ca_system_score_codex":0.0018308347,"about_ca_system_score_gemma":0.0019477672,"threshold_uncertainty_score":0.020905316},"labels":[],"label_agreement":null},{"id":"W4409563182","doi":"10.1007/s13042-025-02622-z","title":"Reward design in multi-agent systems using successor features and multi-information source bayesian optimization","year":2025,"lang":"en","type":"article","venue":"International Journal of Machine Learning and Cybernetics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"National Research Foundation of Korea","keywords":"Successor cardinal; Computational intelligence; Computer science; Bayesian probability; Artificial intelligence; Bayesian optimization; Multi-agent system; Machine learning; Data mining; Mathematics","score_opus":0.016699151340101286,"score_gpt":0.2864372175682467,"score_spread":0.2697380662281454,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409563182","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012183965,0.00025684602,0.9850213,0.0003467132,0.00003748665,0.00006232068,0.000036680554,0.00012264901,0.0019320855],"genre_scores_gemma":[0.884546,0.00028598725,0.11019267,0.0001591346,0.000076900375,0.00034850996,0.00010314735,0.00012404389,0.0041636326],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998384,0.00067056064,0.00010570034,0.00028221804,0.0003464847,0.00021106256],"domain_scores_gemma":[0.9924819,0.0054851645,0.0006274819,0.00016582785,0.0009826618,0.00025685388],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046424386,0.0015340544,0.0035551856,0.0013794565,0.00096440007,0.0024736323,0.0025482979,0.003301131,0.0029270074],"category_scores_gemma":[0.011778498,0.0020283686,0.0011479525,0.0010425117,0.0021228418,0.0030944853,0.0029506637,0.0023650706,0.00040750668],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000057681882,0.000028136285,0.00014886408,0.000055703735,0.000028007942,0.000034516357,0.000032876116,0.9859719,0.00025234162,0.008820537,0.00018290106,0.0043864325],"study_design_scores_gemma":[0.000008997061,0.0000124655735,0.000022687698,0.0000042212837,0.000004048214,0.0000028401616,0.0000020161553,0.9970925,0.00005306715,0.0027494614,0.0000443679,0.0000033594595],"about_ca_topic_score_codex":0.005932585,"about_ca_topic_score_gemma":0.0053261085,"teacher_disagreement_score":0.005932585,"about_ca_system_score_codex":0.0019524123,"about_ca_system_score_gemma":0.002208036,"threshold_uncertainty_score":0.024551868},"labels":[],"label_agreement":null},{"id":"W4409632503","doi":"10.1016/b978-0-443-14081-5.00070-2","title":"Multi-Agent Reinforcement Learning Under General Information Structures","year":2025,"lang":"en","type":"book-chapter","venue":"Elsevier eBooks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Reinforcement learning; Reinforcement; Computer science; Psychology; Artificial intelligence; Social psychology","score_opus":0.018375270866169687,"score_gpt":0.24956501738888678,"score_spread":0.2311897465227171,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409632503","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012739282,0.002927662,0.9147192,0.00064173335,0.00026377154,0.00004007989,0.00007109492,0.00060564524,0.06799145],"genre_scores_gemma":[0.6363332,0.004759288,0.23263326,0.0002501062,0.0003186834,0.00021632035,0.00028518823,0.00024350536,0.12496037],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9998628,0.00003731236,0.0000070271435,0.000026177268,0.0000540543,0.000012599389],"domain_scores_gemma":[0.99968946,0.0002069037,0.000023559985,0.000029096056,0.00003650577,0.0000145144295],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00034699103,0.00065358257,0.00054144644,0.0002459506,0.00014527793,0.0008199622,0.0006830902,0.00075763214,0.0065119653],"category_scores_gemma":[0.0012330131,0.00025084693,0.00028886658,0.00040218068,0.00057345454,0.0009389525,0.0006875428,0.0011967609,0.0011161515],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000056734767,0.00005688077,0.00018495556,0.00018479387,0.000042922347,0.000088658475,0.000045918627,0.64492494,0.0034504605,0.10586543,0.007774198,0.23732401],"study_design_scores_gemma":[0.00001057743,0.000023480283,0.00012357126,0.000024680874,0.0000062899653,0.000029407032,0.0000060570937,0.9331057,0.00064938597,0.05992178,0.0060928552,0.0000062237978],"about_ca_topic_score_codex":0.0014795391,"about_ca_topic_score_gemma":0.0013340743,"teacher_disagreement_score":0.0065119653,"about_ca_system_score_codex":0.0005741521,"about_ca_system_score_gemma":0.00038369137,"threshold_uncertainty_score":0.021784663},"labels":[],"label_agreement":null},{"id":"W4409738260","doi":"10.1016/j.eswa.2025.127717","title":"CCMA: A framework for cascading cooperative multi-agent in autonomous driving merging using Large Language Models","year":2025,"lang":"en","type":"article","venue":"Expert Systems with Applications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canada Research Chairs","funders":"Hunan Provincial Key Laboratory of Materials Protection for Electric Power and Transportation, Changsha University of Science and Technology; Tsinghua Shenzhen International Graduate School; Science, Technology and Innovation Commission of Shenzhen Municipality","keywords":"Computer science; Artificial intelligence","score_opus":0.028189147063689083,"score_gpt":0.32765445691081624,"score_spread":0.29946530984712716,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409738260","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0039139874,0.00010258169,0.99242204,0.000094049814,0.000044829412,0.00005666282,0.000061113,0.0018777326,0.0014269195],"genre_scores_gemma":[0.4080154,0.00020274073,0.5842418,0.0001589094,0.000057262347,0.00048319603,0.00030436207,0.0006107821,0.0059255036],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99944264,0.00017027537,0.000030189916,0.00013890595,0.00014463325,0.00007342348],"domain_scores_gemma":[0.99901223,0.00044243166,0.00007149688,0.00015338877,0.00019437535,0.00012614876],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001310901,0.0009366762,0.0010876459,0.00069490855,0.0012566211,0.0013371649,0.0038857074,0.0016876694,0.0063470905],"category_scores_gemma":[0.0031224801,0.0008031119,0.0012701307,0.0005913393,0.001100979,0.0017695188,0.0037344655,0.0021610302,0.0011809483],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009249812,0.00008580866,0.00040648514,0.00010697213,0.00009048492,0.00019742982,0.00016632475,0.89069873,0.0034934988,0.042734925,0.0032212406,0.05870553],"study_design_scores_gemma":[0.0000047341073,0.00000828243,0.000014490968,0.000002612984,0.0000044349017,0.000008997003,0.0000064586307,0.9924972,0.0002994673,0.006426905,0.0007218955,0.0000045547267],"about_ca_topic_score_codex":0.018333623,"about_ca_topic_score_gemma":0.023922576,"teacher_disagreement_score":0.018333623,"about_ca_system_score_codex":0.0011002488,"about_ca_system_score_gemma":0.002449516,"threshold_uncertainty_score":0.036453784},"labels":[],"label_agreement":null},{"id":"W4409882931","doi":"10.1109/tai.2025.3564900","title":"Prescribed Performance Resilient Motion Coordination With Actor–Critic Reinforcement Learning Design for UAV-USV Systems","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Reinforcement learning; Motion (physics); Reinforcement; Computer science; Aeronautics; Engineering; Artificial intelligence; Structural engineering","score_opus":0.04943525313310071,"score_gpt":0.2841132150108918,"score_spread":0.2346779618777911,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409882931","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018029133,0.00045314748,0.9759424,0.00020032405,0.00007300106,0.000072225885,0.000020359856,0.00017595093,0.005033479],"genre_scores_gemma":[0.9675968,0.00027080043,0.029121071,0.00007239614,0.000036883914,0.00021884331,0.000033133107,0.000028505812,0.0026214572],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993793,0.00017899649,0.000031535004,0.00014532132,0.00017541555,0.00008943267],"domain_scores_gemma":[0.9991345,0.00033359884,0.00018556135,0.00004288936,0.00025133498,0.00005222336],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013764093,0.0014281316,0.0010725423,0.00038135223,0.00038576467,0.001081295,0.0012064988,0.0012214449,0.0014906299],"category_scores_gemma":[0.0017300384,0.0004823318,0.0006164452,0.00027520276,0.0011606822,0.0004894415,0.001308812,0.0010721975,0.0002503853],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003214551,0.000016580123,0.00018766386,0.00005720745,0.000022202508,0.000078555444,0.00004849451,0.9890315,0.0015362181,0.0039014437,0.0001896749,0.004898293],"study_design_scores_gemma":[0.000005225217,0.000030774954,0.000024904451,0.0000033713447,0.0000036931335,0.0000040516675,0.0000036757403,0.9992411,0.00014950738,0.00039111063,0.00014046353,0.0000020590621],"about_ca_topic_score_codex":0.004696604,"about_ca_topic_score_gemma":0.0028386265,"teacher_disagreement_score":0.004696604,"about_ca_system_score_codex":0.00091497716,"about_ca_system_score_gemma":0.0011027182,"threshold_uncertainty_score":0.009338558},"labels":[],"label_agreement":null},{"id":"W4409989961","doi":"10.1101/2025.04.28.651014","title":"Interaction between Model-based and Model-free Mechanisms in Motor Learning","year":2025,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Department of Science and Technology, Ministry of Science and Technology, India","keywords":"Computer science; Motor learning; Psychology; Cognitive science; Neuroscience","score_opus":0.01913034100641843,"score_gpt":0.23865849655027113,"score_spread":0.2195281555438527,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409989961","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8510933,0.00029606937,0.14537805,0.00024123631,0.000040486048,0.00004623321,0.00003994757,0.0005854223,0.0022791536],"genre_scores_gemma":[0.99217004,0.00004729102,0.0074222162,0.00001840353,0.0000040176747,0.0000232677,0.00001757052,0.00003056764,0.0002666359],"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99903035,0.00038010383,0.00006989509,0.00020814428,0.00020201702,0.00010951563],"domain_scores_gemma":[0.99569315,0.0022936915,0.00059993565,0.00090285327,0.0002814652,0.000229044],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020954062,0.0005247879,0.0005838817,0.00026477603,0.00021649801,0.0013467311,0.0008329414,0.0008194966,0.0015043233],"category_scores_gemma":[0.008274575,0.00045428926,0.0006614799,0.00015971763,0.0010071871,0.0018351288,0.0014468865,0.0011937557,0.00021484574],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012398851,0.0011560627,0.030730192,0.00071772054,0.00076354045,0.00031882603,0.0011380279,0.20707254,0.6125186,0.028482841,0.00048346492,0.11537829],"study_design_scores_gemma":[0.00008135723,0.001169416,0.022902012,0.000055170567,0.00019374977,0.0002354156,0.0001308859,0.825688,0.10528979,0.04265586,0.0014770093,0.00012134986],"about_ca_topic_score_codex":0.0005949487,"about_ca_topic_score_gemma":0.00056919834,"teacher_disagreement_score":0.0020954062,"about_ca_system_score_codex":0.0005453816,"about_ca_system_score_gemma":0.0005473822,"threshold_uncertainty_score":0.011081696},"labels":[],"label_agreement":null},{"id":"W4410235858","doi":"10.3390/jcp5020023","title":"Combining Supervised and Reinforcement Learning to Build a Generic Defensive Cyber Agent","year":2025,"lang":"en","type":"article","venue":"Journal of Cybersecurity and Privacy","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Reinforcement learning; Reinforcement; Computer science; Artificial intelligence; Psychology; Social psychology","score_opus":0.017410837487859866,"score_gpt":0.26187635266903847,"score_spread":0.2444655151811786,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410235858","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034890767,0.00032786978,0.9597253,0.0002392318,0.000059305912,0.00010519068,0.000035266235,0.0012927161,0.0033244009],"genre_scores_gemma":[0.8514787,0.00017363667,0.14502,0.0002541299,0.00006003511,0.00015225407,0.00012997509,0.00008273916,0.0026486523],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99947447,0.00013094561,0.00002957496,0.00014677079,0.00014119102,0.000077123645],"domain_scores_gemma":[0.99877816,0.00051360176,0.00019274949,0.00015425167,0.00027422776,0.00008697285],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011773586,0.0011153776,0.0007957254,0.0005034001,0.00032753276,0.00063054776,0.0015200713,0.0010722113,0.0012333093],"category_scores_gemma":[0.0026240603,0.00042683317,0.00053886266,0.00024932972,0.0010565374,0.0009799555,0.0011242898,0.0015098742,0.00037297586],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000030920794,0.000062704574,0.0011179973,0.00004226331,0.00003976921,0.00005231764,0.00003910425,0.95868915,0.0020615468,0.002970193,0.00047283305,0.034421153],"study_design_scores_gemma":[0.0000032481591,0.000024890363,0.00006619736,0.0000033406666,0.0000039679003,0.000008141844,0.0000023957964,0.9984701,0.00040528315,0.00079242716,0.00021750949,0.0000025157703],"about_ca_topic_score_codex":0.005490524,"about_ca_topic_score_gemma":0.005926671,"teacher_disagreement_score":0.005490524,"about_ca_system_score_codex":0.0007684793,"about_ca_system_score_gemma":0.0013592668,"threshold_uncertainty_score":0.010917127},"labels":[],"label_agreement":null},{"id":"W4410431351","doi":"10.1016/j.neunet.2025.107574","title":"Multi-agent self-attention reinforcement learning for multi-USV hunting target","year":2025,"lang":"en","type":"article","venue":"Neural Networks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Nexen (Canada)","funders":"National Major Science and Technology Projects of China; Major Science and Technology Project of Hainan Province; Hainan University; Natural Science Foundation of Hainan Province; National Natural Science Foundation of China","keywords":"Reinforcement learning; Reinforcement; Computer science; Artificial intelligence; Psychology; Social psychology","score_opus":0.025581076575169767,"score_gpt":0.2818252087779485,"score_spread":0.2562441322027787,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410431351","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1183992,0.00059144606,0.8724822,0.00042121825,0.00013610603,0.00007378239,0.00004206432,0.0005171457,0.0073369285],"genre_scores_gemma":[0.97906417,0.00006636161,0.017953906,0.00007946195,0.000019365556,0.0000604687,0.000027361062,0.000022168148,0.0027067726],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998179,0.000037313028,0.000008743961,0.00004707691,0.000044850425,0.000044170163],"domain_scores_gemma":[0.99945086,0.00023012732,0.000081705766,0.00003625906,0.00014565488,0.000055250417],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006677576,0.000640886,0.00085255574,0.00036695146,0.00048742438,0.00053538976,0.0012300735,0.0011332792,0.0019617893],"category_scores_gemma":[0.0016594802,0.00038072802,0.00035507724,0.00025513337,0.0006106243,0.0006276811,0.0012479614,0.00082672265,0.00023146629],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006147571,0.00005755886,0.0005297534,0.000032028023,0.000030172854,0.00006367346,0.000036694648,0.9704658,0.0019940217,0.00268889,0.0006780953,0.023361972],"study_design_scores_gemma":[0.0000031245638,0.000012715189,0.000048992417,0.0000012615931,0.0000022108395,0.000004459779,0.000001971845,0.999335,0.00012528634,0.00040130145,0.00006230009,0.0000013500987],"about_ca_topic_score_codex":0.008055352,"about_ca_topic_score_gemma":0.0063082343,"teacher_disagreement_score":0.008055352,"about_ca_system_score_codex":0.00070398144,"about_ca_system_score_gemma":0.00074322254,"threshold_uncertainty_score":0.01601696},"labels":[],"label_agreement":null},{"id":"W4410537008","doi":"10.1007/s00521-025-11288-1","title":"Human-AI collaboration in real-world complex environment with reinforcement learning","year":2025,"lang":"en","type":"article","venue":"Neural Computing and Applications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Thales (Canada); Espace pour la vie; Network for Business Sustainability; University of Alberta","funders":"","keywords":"Computational Science and Engineering; Reinforcement learning; Computer science; Reinforcement; Artificial intelligence; Machine learning; Psychology; Social psychology","score_opus":0.019318648162714565,"score_gpt":0.3100600253669215,"score_spread":0.29074137720420695,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410537008","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28592482,0.00028475642,0.7037283,0.0006296011,0.00009791481,0.000112756024,0.000030725787,0.0005752674,0.008615831],"genre_scores_gemma":[0.9785452,0.00003331212,0.02041517,0.00002922219,0.000008707123,0.00003151781,0.000010491084,0.000013912511,0.00091234],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99920636,0.00038775217,0.000024676392,0.0001643016,0.00010691718,0.000109931345],"domain_scores_gemma":[0.9976636,0.0012063241,0.00023128642,0.00025593286,0.00024398192,0.00039889204],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016947687,0.00046043843,0.0006385627,0.00033240902,0.0006807363,0.0009902376,0.0011400176,0.00086416554,0.002050564],"category_scores_gemma":[0.004725123,0.0002592556,0.00031007084,0.00025858366,0.0011163232,0.001582281,0.0024342083,0.0009652086,0.00025342323],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059878686,0.000511141,0.0052722306,0.00011466427,0.00015292833,0.000512874,0.0008792265,0.8886016,0.008015586,0.019146351,0.0014384254,0.07475617],"study_design_scores_gemma":[0.000018799281,0.00007268419,0.00034498004,0.0000038582652,0.000008849809,0.00002929454,0.000086017724,0.99017525,0.00065153884,0.008253828,0.0003477184,0.00000718425],"about_ca_topic_score_codex":0.0026974857,"about_ca_topic_score_gemma":0.0023644813,"teacher_disagreement_score":0.0026974857,"about_ca_system_score_codex":0.0005543488,"about_ca_system_score_gemma":0.0009943254,"threshold_uncertainty_score":0.008962929},"labels":[],"label_agreement":null},{"id":"W4410629910","doi":"10.1016/j.neucom.2025.130470","title":"SAGE: Self-evolving Agents with Reflective and Memory-augmented Abilities","year":2025,"lang":"en","type":"article","venue":"Neurocomputing","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Science and Technology Planning Project of Shenzhen Municipality","keywords":"Computer science; SAGE; Artificial intelligence; Cognitive science; Cognitive psychology; Psychology","score_opus":0.00843311187358675,"score_gpt":0.2504298560117046,"score_spread":0.24199674413811784,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410629910","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.095702164,0.00026678474,0.87919927,0.00040916202,0.0002959381,0.00014471426,0.00018098946,0.0072526014,0.016548354],"genre_scores_gemma":[0.71647114,0.00021663858,0.26287234,0.00020293555,0.000041958552,0.00026603288,0.00029686475,0.0003347985,0.0192974],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99989164,0.00002703364,0.0000070512924,0.000021815229,0.00003906638,0.000013326825],"domain_scores_gemma":[0.9996264,0.00015860316,0.000030678144,0.00006949678,0.000053093598,0.00006168329],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00030644576,0.00035710886,0.0003531008,0.00019356095,0.00023691927,0.00055875705,0.0008410885,0.00067143585,0.0038305344],"category_scores_gemma":[0.0014761679,0.0001736938,0.000309032,0.00012432123,0.00055739586,0.00065540173,0.0013051025,0.00077003,0.0010027398],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048105157,0.00034757383,0.0031727066,0.00028701956,0.00016342514,0.00060510315,0.00043092205,0.6414083,0.051000983,0.09485721,0.01236924,0.19487642],"study_design_scores_gemma":[0.0000893225,0.00015019126,0.00028673102,0.000016851087,0.00002780809,0.00014811356,0.00003838135,0.9487653,0.010103314,0.03040214,0.0099553745,0.000016542583],"about_ca_topic_score_codex":0.0005904461,"about_ca_topic_score_gemma":0.000975825,"teacher_disagreement_score":0.0038305344,"about_ca_system_score_codex":0.00017079753,"about_ca_system_score_gemma":0.00039611646,"threshold_uncertainty_score":0.012814403},"labels":[],"label_agreement":null},{"id":"W4410742988","doi":"10.1007/978-3-031-91524-6_1","title":"Why AI and Security?","year":2025,"lang":"en","type":"book-chapter","venue":"Progress in IS","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada; McGill University; MacEwan University; York University; University of Toronto","funders":"","keywords":"Computer science","score_opus":0.01238897226127554,"score_gpt":0.2654781307917201,"score_spread":0.25308915853044456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410742988","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0017981025,0.056971468,0.025615208,0.07357674,0.004080076,0.000022445178,0.00008041393,0.0002725827,0.837583],"genre_scores_gemma":[0.16758467,0.06699935,0.016639415,0.01897096,0.005066394,0.00013980226,0.00018907721,0.0003854676,0.72402483],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99948514,0.00018271898,0.000012447661,0.00007904149,0.00018500297,0.000055641918],"domain_scores_gemma":[0.99929786,0.00040290251,0.00003298157,0.00009005871,0.00011632909,0.000059753816],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00091138505,0.00063864706,0.0005268483,0.00063968654,0.0011190333,0.004900248,0.0006661704,0.0023314774,0.022832874],"category_scores_gemma":[0.0021104268,0.00029097058,0.00019915085,0.0007635779,0.007443323,0.009493982,0.0013237101,0.005359616,0.007502936],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000047364742,0.000008150449,0.000027072285,0.000055041848,0.0000022773584,0.000009069204,0.00016774565,0.0002306584,0.00007483071,0.91730356,0.051008247,0.031108592],"study_design_scores_gemma":[0.000003185675,0.000006993153,0.00006587864,0.00012756836,0.0000017540766,0.000037602946,0.00019064508,0.00051193044,0.000108900735,0.5504126,0.44852754,0.000005325826],"about_ca_topic_score_codex":0.0021301056,"about_ca_topic_score_gemma":0.002140523,"teacher_disagreement_score":0.022832874,"about_ca_system_score_codex":0.0018819997,"about_ca_system_score_gemma":0.001313799,"threshold_uncertainty_score":0.07638359},"labels":[],"label_agreement":null},{"id":"W4410779232","doi":"10.2139/ssrn.5270695","title":"Decentralized Security for Multi-User Vr Applications Using Multi-Agent Reinforcement Learning","year":2025,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Reinforcement learning; Computer science; Human–computer interaction; Reinforcement; Artificial intelligence; Engineering","score_opus":0.038889901537594175,"score_gpt":0.33001636359697195,"score_spread":0.29112646205937776,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410779232","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.058014557,0.0001317691,0.9375796,0.00018900703,0.00004117341,0.00009640231,0.000021581132,0.0009178807,0.0030080352],"genre_scores_gemma":[0.96773225,0.00003458105,0.03067933,0.000030988856,0.000010097173,0.00005764083,0.000017729255,0.000038107606,0.0013992385],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99920386,0.0002470738,0.00003143468,0.00015630436,0.000201062,0.00016026177],"domain_scores_gemma":[0.9981243,0.0008839812,0.00025890578,0.00028855194,0.00028839128,0.00015587422],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010576054,0.000611435,0.00095642597,0.0003245466,0.0006018947,0.00097246334,0.0010415462,0.00094086496,0.002901459],"category_scores_gemma":[0.0033867853,0.00034061715,0.00047745137,0.00021399384,0.00091377355,0.0012140984,0.0019613889,0.0014480443,0.00036480807],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038760505,0.00017680807,0.00072459644,0.00008415656,0.00005329468,0.00020467283,0.00015409771,0.9087286,0.015780983,0.013166744,0.0011436003,0.05939483],"study_design_scores_gemma":[0.0000120654195,0.00003542205,0.00008217005,0.0000026723728,0.0000033179144,0.000016288246,0.0000085840065,0.9963528,0.0007614585,0.0025591892,0.00016180838,0.0000041880808],"about_ca_topic_score_codex":0.0025303264,"about_ca_topic_score_gemma":0.002363414,"teacher_disagreement_score":0.002901459,"about_ca_system_score_codex":0.0007065116,"about_ca_system_score_gemma":0.00089734315,"threshold_uncertainty_score":0.009706378},"labels":[],"label_agreement":null},{"id":"W4410887809","doi":"10.1109/syscon64521.2025.11014852","title":"A Simulation Pipeline to Facilitate Real-World Robotic Reinforcement Learning Applications","year":2025,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Reinforcement learning; Pipeline (software); Computer science; Human–computer interaction; Artificial intelligence; Operating system","score_opus":0.0397107984912193,"score_gpt":0.3063709812696117,"score_spread":0.26666018277839243,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410887809","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008582418,0.000053394542,0.96915466,0.00017124359,0.00004787963,0.00036922254,0.00017096748,0.015954848,0.005495393],"genre_scores_gemma":[0.30568245,0.00021848362,0.6854272,0.00014856609,0.0000217135,0.000976754,0.00085968454,0.0011044507,0.0055607185],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99958605,0.00012620508,0.000033320306,0.000076762335,0.00012483734,0.000052762556],"domain_scores_gemma":[0.99831057,0.0008747519,0.00008920499,0.00026661114,0.00027710065,0.00018177436],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012803844,0.0009999083,0.00047968343,0.00060990575,0.00038102048,0.00069793,0.0017740849,0.0008475932,0.0167151],"category_scores_gemma":[0.0033708408,0.0007163791,0.0005760967,0.000264291,0.00059565593,0.0010906549,0.0019675312,0.001708906,0.0026375742],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007441055,0.0008281885,0.0022222789,0.00034485583,0.000059979615,0.0005024161,0.00051285065,0.7591223,0.040938582,0.021631474,0.012294087,0.16079897],"study_design_scores_gemma":[0.000084965126,0.0001366201,0.00023185494,0.000025119141,0.000008907866,0.000053386266,0.000021321672,0.9715343,0.009435292,0.00392839,0.014518939,0.000020875264],"about_ca_topic_score_codex":0.0035769665,"about_ca_topic_score_gemma":0.0031923132,"teacher_disagreement_score":0.0167151,"about_ca_system_score_codex":0.0006457991,"about_ca_system_score_gemma":0.0017111029,"threshold_uncertainty_score":0.05591762},"labels":[],"label_agreement":null},{"id":"W4410988898","doi":"10.1016/j.trc.2025.105183","title":"Curiosity-driven reinforcement learning with graph transformers for decision-making in connected and autonomous vehicles","year":2025,"lang":"en","type":"article","venue":"Transportation Research Part C Emerging Technologies","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Curiosity; Reinforcement learning; Transformer; Reinforcement; Graph; Engineering; Computer science; Artificial intelligence; Psychology; Electrical engineering; Theoretical computer science; Social psychology; Structural engineering","score_opus":0.027220519970360464,"score_gpt":0.3392383197203113,"score_spread":0.31201779974995086,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410988898","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07764527,0.00013073774,0.9180165,0.00027999317,0.000053102896,0.000083729465,0.00007818643,0.00032414362,0.0033883718],"genre_scores_gemma":[0.962496,0.000060439164,0.03594299,0.000052235602,0.000010274653,0.00007195065,0.000042359934,0.00003275914,0.00129091],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996025,0.00016790377,0.000017121381,0.00008622617,0.000062077015,0.00006416677],"domain_scores_gemma":[0.9963624,0.002831487,0.00018967231,0.00015489639,0.00025277355,0.00020878842],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010887821,0.00049463386,0.00078174035,0.00043656427,0.00034848944,0.00074658124,0.0011898663,0.00071410625,0.0033273974],"category_scores_gemma":[0.00655109,0.00032570434,0.0004833466,0.00031317244,0.0012535164,0.0013752031,0.0013134335,0.0015425817,0.00022145272],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001997028,0.00012487992,0.0007261771,0.00008452554,0.00004864232,0.00007294505,0.00014532998,0.8929305,0.001694682,0.06673956,0.00078626635,0.03644675],"study_design_scores_gemma":[0.000014161795,0.000023699473,0.000042652664,0.0000033996087,0.0000048728084,0.000004811448,0.000007402173,0.9670055,0.00019349283,0.032609817,0.00008671437,0.000003510753],"about_ca_topic_score_codex":0.0041347374,"about_ca_topic_score_gemma":0.004596998,"teacher_disagreement_score":0.0041347374,"about_ca_system_score_codex":0.0009249031,"about_ca_system_score_gemma":0.0011816117,"threshold_uncertainty_score":0.011131287},"labels":[],"label_agreement":null},{"id":"W4411173295","doi":"10.1109/icse-nier66352.2025.00008","title":"Listening to the Firehose: Sonifying Z3’s Behavior","year":2025,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Active listening; Computer science; Speech recognition; Psychology; Communication","score_opus":0.014707942629321126,"score_gpt":0.2865948154057638,"score_spread":0.2718868727764427,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411173295","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27984384,0.00030373,0.64383686,0.00090078387,0.00016853656,0.00022816744,0.0027326394,0.054702748,0.01728271],"genre_scores_gemma":[0.8398937,0.00018037725,0.14804022,0.0004154053,0.000028566674,0.00015413218,0.0017556453,0.0034839092,0.0060480842],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9994679,0.00010901942,0.00002483368,0.000110663066,0.00021434475,0.00007334429],"domain_scores_gemma":[0.9981229,0.0011521732,0.0001252938,0.00033819166,0.00016937027,0.00009220128],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010337697,0.00062462746,0.00033280885,0.00057079917,0.00041149548,0.0009470621,0.0011260235,0.00060430204,0.009311275],"category_scores_gemma":[0.006587799,0.00029643803,0.00037522105,0.000379909,0.0010340298,0.0015997102,0.0017725764,0.0008192808,0.0017235925],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0040159854,0.00028589292,0.051983118,0.0008989078,0.00018279302,0.0028182385,0.0076387087,0.08155006,0.19784115,0.051144555,0.05888564,0.5427549],"study_design_scores_gemma":[0.0002198646,0.0004738415,0.017170537,0.00015290102,0.00008266527,0.00096000097,0.0023142376,0.75848675,0.10661994,0.05120088,0.06213841,0.00017999759],"about_ca_topic_score_codex":0.0039424547,"about_ca_topic_score_gemma":0.005727764,"teacher_disagreement_score":0.009311275,"about_ca_system_score_codex":0.00043775173,"about_ca_system_score_gemma":0.0004748339,"threshold_uncertainty_score":0.031149387},"labels":[],"label_agreement":null},{"id":"W4411225534","doi":"10.1017/9781009504942.008","title":"Reinforcement learning","year":2025,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; University of Waterloo; McMaster University","funders":"","keywords":"Reinforcement; Reinforcement learning; Psychology; Computer science; Artificial intelligence; Social psychology","score_opus":0.018228091514921805,"score_gpt":0.20180634372568926,"score_spread":0.18357825221076746,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411225534","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0037262235,0.0056670923,0.90036017,0.0026953674,0.0006986959,0.00014196281,0.0002612346,0.000683782,0.0857655],"genre_scores_gemma":[0.42049238,0.018023042,0.41734308,0.0026688918,0.0011497056,0.0011818318,0.0013089655,0.00045307248,0.13737907],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9993568,0.0002422068,0.000035527086,0.00013830753,0.0001814521,0.000045737474],"domain_scores_gemma":[0.9989785,0.00067720073,0.00005226842,0.00009855881,0.00013422687,0.00005930101],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009740457,0.0010877381,0.00082166103,0.00040309998,0.000401485,0.0015211675,0.0013569135,0.0011718035,0.018702764],"category_scores_gemma":[0.0038756302,0.00027469054,0.00052768283,0.00049215794,0.0011837888,0.0014605646,0.0011727973,0.0018332249,0.0039315023],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006303934,0.000117204254,0.00054721057,0.00045137983,0.000083169965,0.000098723554,0.00011038265,0.16568087,0.0011530747,0.6063812,0.02592192,0.19939184],"study_design_scores_gemma":[0.000056278182,0.0000952078,0.00022576997,0.00016948524,0.000028476094,0.00012206912,0.000051703293,0.30292234,0.0008820318,0.5864353,0.10897964,0.000031700314],"about_ca_topic_score_codex":0.0011427068,"about_ca_topic_score_gemma":0.0014377828,"teacher_disagreement_score":0.018702764,"about_ca_system_score_codex":0.0010833436,"about_ca_system_score_gemma":0.0010323506,"threshold_uncertainty_score":0.062566996},"labels":[],"label_agreement":null},{"id":"W4411597815","doi":"10.1101/2025.06.17.659642","title":"Biological Reasoning with Reinforcement Learning through Natural Language Enables Generalizable Zero-Shot Cell Type Annotations","year":2025,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Zero (linguistics); Computer science; Reinforcement; Shot (pellet); One shot; Artificial intelligence; Natural (archaeology); Natural language; Psychology; Linguistics; Biology; Engineering; Chemistry; Social psychology","score_opus":0.01926935263878364,"score_gpt":0.24168362320151324,"score_spread":0.2224142705627296,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411597815","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15996875,0.0008492409,0.7858463,0.0017197351,0.0002581312,0.00020174029,0.004914078,0.039723612,0.006518408],"genre_scores_gemma":[0.71245754,0.0002253571,0.2743643,0.0012219646,0.00006757342,0.0002203297,0.008100614,0.0007258975,0.0026163792],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99873036,0.0003005321,0.00006727293,0.0005331449,0.00026012555,0.00010859118],"domain_scores_gemma":[0.9957273,0.0028709108,0.0002765426,0.0005539591,0.0004069975,0.00016421618],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020190696,0.0011499358,0.00065896084,0.0007469829,0.0004252892,0.0015775934,0.0029209198,0.0013093874,0.0044619236],"category_scores_gemma":[0.0103054745,0.0004941994,0.0013642833,0.00043489016,0.0011133151,0.002386016,0.0018902556,0.0024624844,0.0015966538],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00083245005,0.00066947663,0.0137296375,0.0010509916,0.00026295832,0.000495907,0.0004533177,0.6110136,0.045783643,0.020637486,0.022717204,0.28235328],"study_design_scores_gemma":[0.000026921593,0.000041770047,0.00045981712,0.000020331718,0.00001666627,0.000031922256,0.000031018848,0.97750527,0.0067872945,0.0135798985,0.0014846001,0.000014545273],"about_ca_topic_score_codex":0.007851654,"about_ca_topic_score_gemma":0.009754635,"teacher_disagreement_score":0.007851654,"about_ca_system_score_codex":0.0014552113,"about_ca_system_score_gemma":0.0021243102,"threshold_uncertainty_score":0.015611947},"labels":[],"label_agreement":null},{"id":"W4411745406","doi":"10.1007/978-3-031-95976-9_9","title":"Reinforcement Learning-Based Heuristics to Guide Domain-Independent Dynamic Programming","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"The King's University; Canada Research Chairs; University of Toronto","funders":"","keywords":"Computer science; Reinforcement learning; Heuristics; Domain (mathematical analysis); Artificial intelligence; Dynamic programming; Machine learning; Algorithm; Operating system; Mathematics","score_opus":0.009434525304242686,"score_gpt":0.2577204471215944,"score_spread":0.24828592181735173,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411745406","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006435506,0.00018384584,0.9874263,0.000096991906,0.000045808436,0.00006026386,0.000036630954,0.0006092026,0.005105451],"genre_scores_gemma":[0.4046748,0.00033496864,0.5883189,0.00017919684,0.00005659522,0.0003260112,0.00019970376,0.00041209316,0.0054977806],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9996778,0.00011225826,0.000017877992,0.000071337316,0.00007107798,0.000049738916],"domain_scores_gemma":[0.998398,0.0011593453,0.000081208804,0.00011492277,0.00016636655,0.00008020118],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00092952093,0.0011045339,0.0011020703,0.0006993745,0.00047276204,0.0010118026,0.0015780553,0.0010211855,0.0057824776],"category_scores_gemma":[0.004490439,0.00068579585,0.00061524607,0.00065478176,0.00093184673,0.0011853069,0.0014797935,0.0021178608,0.000957662],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008268053,0.00008783085,0.0002681454,0.00009151831,0.000028814684,0.00005081471,0.00008832467,0.83776754,0.0014803833,0.042748023,0.0036055886,0.11370035],"study_design_scores_gemma":[0.000016259275,0.000014985576,0.000027700378,0.000012515676,0.0000042852334,0.0000070865335,0.000006724551,0.9847809,0.00035912756,0.01402602,0.00073975965,0.0000046141545],"about_ca_topic_score_codex":0.0038003654,"about_ca_topic_score_gemma":0.0065243286,"teacher_disagreement_score":0.0057824776,"about_ca_system_score_codex":0.0009921427,"about_ca_system_score_gemma":0.0014069331,"threshold_uncertainty_score":0.01934433},"labels":[],"label_agreement":null},{"id":"W4411745413","doi":"10.1007/978-3-031-95976-9_16","title":"Shaping Reward Signals in Reinforcement Learning Using Constraint Programming","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Reinforcement learning; Computer science; Constraint (computer-aided design); Artificial intelligence; Constraint programming; Reinforcement; Constraint satisfaction; Mathematical optimization; Psychology; Mathematics; Stochastic programming","score_opus":0.03778212600561777,"score_gpt":0.28212690686802855,"score_spread":0.2443447808624108,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411745413","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0036160958,0.00044487006,0.98597103,0.00015228939,0.00003962328,0.000030515039,0.000025220219,0.00016151048,0.0095588],"genre_scores_gemma":[0.41142344,0.0014179512,0.5685252,0.0002580405,0.00010541745,0.00038966836,0.00012476227,0.0003480548,0.017407535],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9994654,0.00021653461,0.000023753191,0.00009041669,0.0001558925,0.00004816166],"domain_scores_gemma":[0.9982498,0.0014192975,0.000078677884,0.00008096571,0.00012597823,0.000045189696],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010291021,0.0008422514,0.00084931,0.0002900237,0.00034811642,0.0012423812,0.0015083216,0.0010667909,0.0055932445],"category_scores_gemma":[0.00378371,0.00060507143,0.00053706364,0.0009044748,0.0014920074,0.0013478255,0.0011122131,0.0027369459,0.000651271],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000721411,0.000051238017,0.0001194995,0.00017006524,0.000032885062,0.000047577334,0.00006411762,0.7418224,0.002078838,0.15364577,0.002582769,0.0993127],"study_design_scores_gemma":[0.000010752304,0.000016544129,0.000029312376,0.00001733542,0.000005440295,0.000010774569,0.0000057668262,0.94321597,0.0007287978,0.05421097,0.0017399932,0.000008443227],"about_ca_topic_score_codex":0.0034899802,"about_ca_topic_score_gemma":0.003306476,"teacher_disagreement_score":0.0055932445,"about_ca_system_score_codex":0.0011021243,"about_ca_system_score_gemma":0.00090957107,"threshold_uncertainty_score":0.018711269},"labels":[],"label_agreement":null},{"id":"W4412026837","doi":"10.1016/j.neucom.2025.130917","title":"A reinforcement learning-assisted genetic programming algorithm for team formation problem considering person-job matching","year":2025,"lang":"en","type":"article","venue":"Neurocomputing","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"National Natural Science Foundation of China","keywords":"Reinforcement learning; Computer science; Matching (statistics); Genetic algorithm; Artificial intelligence; Genetic programming; Reinforcement; Mathematical optimization; Algorithm; Machine learning; Mathematics; Psychology","score_opus":0.020509647936770344,"score_gpt":0.25831666010610516,"score_spread":0.2378070121693348,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412026837","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023923269,0.0001725247,0.9712433,0.00027049438,0.00009376714,0.000100641024,0.000031311756,0.00026311216,0.0039016658],"genre_scores_gemma":[0.58537906,0.00019310717,0.40862963,0.00026354392,0.00007705661,0.00048307172,0.00012452097,0.000084355044,0.0047656447],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996724,0.00009820435,0.000011650144,0.00007223478,0.00008321296,0.00006225376],"domain_scores_gemma":[0.9993612,0.00035337373,0.00004905692,0.000027846321,0.00014254631,0.00006588826],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00097262405,0.0007032603,0.0013680852,0.00061959296,0.0006583036,0.0006771972,0.0019284133,0.0019850016,0.0029767642],"category_scores_gemma":[0.0019148835,0.00045861266,0.00065530493,0.0006117049,0.00072025403,0.0005435544,0.0012604808,0.0012789767,0.0003863876],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000054892633,0.00008066941,0.0003928468,0.00003127697,0.000027105934,0.0000591899,0.000041914565,0.9530524,0.00075172016,0.0040168106,0.0009473958,0.040543813],"study_design_scores_gemma":[0.000014638147,0.00001805766,0.000029812847,0.0000027182898,0.0000034380428,0.0000071158534,0.00000394077,0.99904424,0.00007159783,0.0006844793,0.00011782296,0.0000020067494],"about_ca_topic_score_codex":0.010302712,"about_ca_topic_score_gemma":0.0058540357,"teacher_disagreement_score":0.010302712,"about_ca_system_score_codex":0.00081156654,"about_ca_system_score_gemma":0.0018711488,"threshold_uncertainty_score":0.02048552},"labels":[],"label_agreement":null},{"id":"W4412614147","doi":"10.21105/joss.07728","title":"EvoVis: Dashboard for Visualizing Evolutionary Neural Architecture Search Algorithms","year":2025,"lang":"en","type":"article","venue":"The Journal of Open Source Software","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Fonds de Recherche du Québec - Santé; Bayerische Forschungsallianz","keywords":"Computer science; Evolutionary algorithm; Architecture; Dashboard; Artificial intelligence; Machine learning; Data mining; Algorithm; Data science; Geography; Archaeology","score_opus":0.03184052293001022,"score_gpt":0.3412038084290155,"score_spread":0.30936328549900527,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412614147","genre_codex":"software","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":"software","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01417432,0.001092622,0.41384286,0.0010905116,0.0009653062,0.00035409397,0.057862718,0.48469564,0.02592194],"genre_scores_gemma":[0.21299069,0.0022427596,0.46709508,0.0017678883,0.00033704974,0.0024878543,0.1403359,0.13568942,0.0370533],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99954164,0.00009068353,0.000043360393,0.000094442104,0.00015607144,0.0000737602],"domain_scores_gemma":[0.9980159,0.0011725315,0.00009320414,0.00020652956,0.00032263287,0.00018928428],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00095455296,0.0027770638,0.0010605904,0.0014354071,0.00050992874,0.0022797142,0.0024411576,0.001513893,0.08699008],"category_scores_gemma":[0.0061379434,0.0007208405,0.0013434789,0.0012117097,0.00052603614,0.0023022362,0.002761169,0.0027733753,0.016117115],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001616699,0.0003357599,0.0042494093,0.0018549613,0.0002366215,0.000800629,0.0011271234,0.07406178,0.008650967,0.03367003,0.7137019,0.1596941],"study_design_scores_gemma":[0.00080589176,0.00022061018,0.0038507844,0.000596184,0.00007271212,0.00036290733,0.00031059142,0.49435696,0.014993743,0.060629267,0.42358664,0.00021369418],"about_ca_topic_score_codex":0.00499348,"about_ca_topic_score_gemma":0.0063635944,"teacher_disagreement_score":0.08699008,"about_ca_system_score_codex":0.0006884089,"about_ca_system_score_gemma":0.0010006079,"threshold_uncertainty_score":0.29101086},"labels":[],"label_agreement":null},{"id":"W4413042885","doi":"10.2139/ssrn.5361139","title":"Symphony: Edge-powered Decentralized Multi-agent Framework for Autonomous Co-evolving Intelligence","year":2025,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Symphony; Enhanced Data Rates for GSM Evolution; Computer science; Artificial intelligence; Art; Art history","score_opus":0.02840843717819469,"score_gpt":0.32375354165604603,"score_spread":0.29534510447785134,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413042885","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010275697,0.00016321104,0.98087734,0.00014612923,0.000092396374,0.000044176784,0.00007740789,0.0013744253,0.0069492417],"genre_scores_gemma":[0.6515249,0.00029093272,0.33450893,0.00016953163,0.000090901136,0.0002921915,0.00024739956,0.00033396823,0.0125412075],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99984634,0.000044291384,0.0000068117124,0.000030386609,0.000043894517,0.000028313145],"domain_scores_gemma":[0.9998293,0.000050393464,0.000014799835,0.000029424455,0.000027345017,0.000048735772],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00037659032,0.00042300415,0.0006347322,0.00025463736,0.00049907697,0.0008643689,0.0014286349,0.00079064537,0.0059739333],"category_scores_gemma":[0.00065005716,0.00024345212,0.00041631368,0.00022907343,0.000610081,0.00085307047,0.0022377658,0.0010377296,0.00090895506],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033522144,0.00010253438,0.0005214058,0.0001889327,0.00008920549,0.00035399603,0.00013958845,0.6488335,0.012684777,0.2074007,0.0097426,0.119607516],"study_design_scores_gemma":[0.000028768336,0.000037453232,0.000051350875,0.0000055993364,0.000007560284,0.000034146564,0.000009609186,0.962462,0.0010675445,0.031927932,0.0043618595,0.0000062685795],"about_ca_topic_score_codex":0.0009482242,"about_ca_topic_score_gemma":0.001553994,"teacher_disagreement_score":0.0059739333,"about_ca_system_score_codex":0.00030766183,"about_ca_system_score_gemma":0.00065578474,"threshold_uncertainty_score":0.019984782},"labels":[],"label_agreement":null},{"id":"W4413180594","doi":"10.1109/isscs66034.2025.11105702","title":"Learning Optimal Agent Behavior From a Synthetic Reasoning Action Dataset","year":2025,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Ontario Ministry of Research, Innovation and Science","keywords":"Computer science; Action (physics); Artificial intelligence; Machine learning","score_opus":0.025904225356853723,"score_gpt":0.3018162174335818,"score_spread":0.27591199207672806,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413180594","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7910072,0.0019176734,0.17470957,0.00230239,0.00032202236,0.000605995,0.010414136,0.0068524284,0.0118686035],"genre_scores_gemma":[0.9191374,0.00022223585,0.06383204,0.00030328767,0.000045610162,0.00029911147,0.013121936,0.00016073175,0.002877667],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989986,0.00038292198,0.000047507685,0.00035201685,0.00012294279,0.00009605468],"domain_scores_gemma":[0.9958176,0.003039394,0.00023181994,0.00037367202,0.00032955737,0.00020793368],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020770233,0.0018988657,0.0008082103,0.0011859717,0.0006058779,0.0012395644,0.0017810295,0.0019693165,0.0031574504],"category_scores_gemma":[0.006801309,0.00056740793,0.0012127616,0.00073337543,0.0008632019,0.001012589,0.0009052947,0.002249913,0.0010353181],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052661356,0.0005518876,0.008296259,0.00023612066,0.00016511255,0.00022139997,0.000110050154,0.9226906,0.0015209793,0.00263508,0.008255884,0.05479014],"study_design_scores_gemma":[0.000041159397,0.0000664254,0.00063266733,0.000013269504,0.00001269136,0.000016163138,0.000024514007,0.9954542,0.0006752809,0.0023500188,0.0007055114,0.000008072352],"about_ca_topic_score_codex":0.01732232,"about_ca_topic_score_gemma":0.029172618,"teacher_disagreement_score":0.01732232,"about_ca_system_score_codex":0.002524208,"about_ca_system_score_gemma":0.0015776332,"threshold_uncertainty_score":0.03444302},"labels":[],"label_agreement":null},{"id":"W4413276903","doi":"10.2514/6.2025-3412.c1","title":"Correction: Hybrid Framework for UAV Mission Planning Using Logic-Based Imaging and Reinforcement Learning","year":2025,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Nexen (Canada)","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence","score_opus":0.03083045445267923,"score_gpt":0.3194727886290114,"score_spread":0.28864233417633217,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413276903","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0052089584,0.000088743596,0.9900557,0.0001404684,0.00014710284,0.000038208535,0.000047852493,0.0019888394,0.0022842083],"genre_scores_gemma":[0.63866055,0.000102102225,0.34925836,0.00024097829,0.000116522926,0.00012775834,0.00020498894,0.00043576892,0.010853081],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99958295,0.00007213374,0.000016573536,0.00009311371,0.0001708729,0.00006432868],"domain_scores_gemma":[0.99925727,0.0002008738,0.00006163329,0.00011395756,0.00031278929,0.000053495864],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006865374,0.0007990226,0.00073326984,0.00050656614,0.000548605,0.000923856,0.0019211401,0.001002671,0.008635462],"category_scores_gemma":[0.0022652475,0.0003574179,0.00051612814,0.00030350796,0.00067396474,0.00090271066,0.0013719884,0.0014123047,0.001174113],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036083075,0.000090610556,0.0006948078,0.00016373953,0.00004618204,0.00021004613,0.00008838124,0.67193586,0.010944285,0.020651164,0.0070259282,0.2877881],"study_design_scores_gemma":[0.000017533843,0.000031413714,0.0000662914,0.0000073303027,0.0000063859734,0.000025678328,0.0000060890657,0.9922144,0.0020999636,0.0041134115,0.0014049191,0.0000065737986],"about_ca_topic_score_codex":0.008334553,"about_ca_topic_score_gemma":0.008380286,"teacher_disagreement_score":0.008635462,"about_ca_system_score_codex":0.0007074751,"about_ca_system_score_gemma":0.001666907,"threshold_uncertainty_score":0.028888524},"labels":[],"label_agreement":null},{"id":"W4413370993","doi":"10.1007/s10458-025-09721-9","title":"Designing policies for transition-independent multiagent systems that are robust to communication loss","year":2025,"lang":"en","type":"article","venue":"Autonomous Agents and Multi-Agent Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Army Research Laboratory; Air Force Research Laboratory; Army Research Office","keywords":"Transition (genetics); Computer science; Multi-agent system; Distributed computing; Artificial intelligence; Chemistry","score_opus":0.06207849552602331,"score_gpt":0.2945134472691957,"score_spread":0.23243495174317239,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413370993","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023162242,0.0001556589,0.9739067,0.00026527,0.00005395036,0.00010169978,0.000029103148,0.00036945954,0.001955889],"genre_scores_gemma":[0.91411823,0.00019778628,0.083021365,0.00017573686,0.000042006202,0.00027065264,0.00007613219,0.0001227034,0.0019754858],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99873286,0.00035712955,0.00008905916,0.00028825435,0.00028029227,0.00025239697],"domain_scores_gemma":[0.9940555,0.0035822545,0.0009181711,0.0004207869,0.00069084804,0.00033240224],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024700994,0.0013525673,0.0012696129,0.00073971815,0.000714843,0.0013990313,0.0015912502,0.0017119999,0.0016628229],"category_scores_gemma":[0.011756294,0.0009662874,0.00067121064,0.0004636885,0.0015829215,0.00173056,0.0025937841,0.002481741,0.0005220297],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012263547,0.00006611555,0.0004623902,0.000094695424,0.000050673465,0.00010291665,0.00014268667,0.957977,0.0038186687,0.01818707,0.00069632014,0.01827882],"study_design_scores_gemma":[0.000020104244,0.00003375463,0.00006172196,0.000009153151,0.000007239302,0.000012725124,0.000018239707,0.9912509,0.00067154836,0.00769392,0.00021535257,0.0000053742956],"about_ca_topic_score_codex":0.0023357926,"about_ca_topic_score_gemma":0.0016487231,"teacher_disagreement_score":0.0024700994,"about_ca_system_score_codex":0.0010862434,"about_ca_system_score_gemma":0.0020041333,"threshold_uncertainty_score":0.013063312},"labels":[],"label_agreement":null},{"id":"W4413479704","doi":"10.36227/techrxiv.175606815.51615643/v1","title":"Real-Time Adaptive Loss Functions for Generative Models Using Reinforcement Learning and Meta-Learning","year":2025,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Reinforcement learning; Meta learning (computer science); Generative grammar; Computer science; Reinforcement; Generative model; Artificial intelligence; Machine learning; Meta-analysis; Adaptive learning; Psychology; Engineering; Social psychology","score_opus":0.054915963465651696,"score_gpt":0.29324934567146044,"score_spread":0.23833338220580874,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413479704","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03311185,0.00019082759,0.96373636,0.00022642629,0.00003943922,0.000062985535,0.000029673769,0.0014492623,0.0011532288],"genre_scores_gemma":[0.8537524,0.00010462425,0.14379296,0.00020703555,0.00003455002,0.00016728541,0.00007139846,0.00021911654,0.0016507339],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994048,0.0002070686,0.00002937356,0.00015005587,0.00013407583,0.00007460458],"domain_scores_gemma":[0.9977113,0.0013701067,0.00022670334,0.0003634218,0.00022275354,0.0001057538],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025663904,0.0012839711,0.00090469024,0.00047378294,0.00031255532,0.0010813731,0.0025832865,0.0011853927,0.0014053368],"category_scores_gemma":[0.0066836895,0.00066570647,0.0006866604,0.00032490262,0.0012934986,0.001887193,0.0014176158,0.0030337234,0.00045754836],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000071259994,0.00007630928,0.0008275987,0.000032092357,0.00005608837,0.00003802211,0.000044596767,0.9447767,0.003724372,0.004034919,0.00049101404,0.04582702],"study_design_scores_gemma":[0.0000047419917,0.000018708475,0.000040898118,0.0000027053966,0.0000037382033,0.000007465495,0.000002103931,0.99729234,0.0009086326,0.001627139,0.00008837769,0.0000031529003],"about_ca_topic_score_codex":0.0030901812,"about_ca_topic_score_gemma":0.0038913111,"teacher_disagreement_score":0.0030901812,"about_ca_system_score_codex":0.0014130352,"about_ca_system_score_gemma":0.0008322412,"threshold_uncertainty_score":0.013572574},"labels":[],"label_agreement":null},{"id":"W4413513049","doi":"10.1016/j.engappai.2025.112037","title":"An adaptive reinforcement learning approach with trait-awareness for heterogeneous multi-robot cooperative pursuit","year":2025,"lang":"en","type":"article","venue":"Engineering Applications of Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"Natural Science Foundation of Shandong Province; National Natural Science Foundation of China","keywords":"Computer science; Reinforcement learning; Robot; Trait; Artificial intelligence; Machine learning; Human–computer interaction","score_opus":0.036737996253077866,"score_gpt":0.2972914045778988,"score_spread":0.26055340832482093,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413513049","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022580193,0.00011140104,0.9749888,0.00013328294,0.000039252845,0.00003329891,0.000010611715,0.00013887044,0.0019642573],"genre_scores_gemma":[0.9350672,0.000080145364,0.06268691,0.00008253873,0.00003905705,0.000112685244,0.00002381631,0.000030690044,0.0018770355],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99957055,0.00011856638,0.000022335358,0.00010103697,0.00011583658,0.00007176636],"domain_scores_gemma":[0.99914145,0.000440598,0.000114475886,0.00008143065,0.00014865639,0.00007343762],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009797609,0.00065641163,0.0010883517,0.00041591562,0.00050868304,0.0006861158,0.0017915572,0.0010417799,0.0014174519],"category_scores_gemma":[0.0024037277,0.00037884768,0.00063042337,0.00038841975,0.0009388935,0.0008460529,0.001965985,0.0010705619,0.0002188138],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009144792,0.00008650018,0.00062772137,0.00005033822,0.00007719532,0.00011689398,0.000090902555,0.93998116,0.0042864233,0.017977213,0.0005901501,0.036024056],"study_design_scores_gemma":[0.0000062510835,0.000019206349,0.000039118448,0.0000013109969,0.0000050301537,0.000007413765,0.0000029880453,0.9977271,0.000121319,0.0019880857,0.00007963908,0.0000026393611],"about_ca_topic_score_codex":0.0030330068,"about_ca_topic_score_gemma":0.0022053206,"teacher_disagreement_score":0.0030330068,"about_ca_system_score_codex":0.0006057497,"about_ca_system_score_gemma":0.00074881216,"threshold_uncertainty_score":0.0060307384},"labels":[],"label_agreement":null},{"id":"W4414128997","doi":"10.1007/978-3-032-04558-4_8","title":"Learning to Optimize Entropy in the Soft Actor-Critic","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Reinforcement learning; Benchmarking; Regularization (linguistics); Entropy (arrow of time); Artificial neural network; Simulated annealing; Source code","score_opus":0.013576386183217663,"score_gpt":0.2525143598403673,"score_spread":0.23893797365714964,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414128997","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0148471445,0.00044157752,0.97940296,0.00033657646,0.00008882636,0.000026440128,0.00003474548,0.00025994654,0.0045616534],"genre_scores_gemma":[0.84419245,0.00042816743,0.14170468,0.00030534537,0.00021152008,0.0002006947,0.00011121097,0.00035005217,0.012495929],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99947566,0.00022149562,0.000025723693,0.00009586924,0.00010838582,0.000072836854],"domain_scores_gemma":[0.99762696,0.0018748597,0.00011983252,0.00010371342,0.00018022991,0.00009437678],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018733516,0.0012410418,0.0016210775,0.00055975764,0.00042308826,0.0013244565,0.0014370087,0.001582034,0.0026543164],"category_scores_gemma":[0.005687859,0.0008120659,0.00048716122,0.0005292607,0.0018446363,0.0016058458,0.0021410997,0.002314308,0.0005754026],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000659952,0.000023351799,0.00019284866,0.000069164686,0.000039301318,0.000029301507,0.000031224834,0.94884586,0.00084209826,0.03061875,0.0009394504,0.018302754],"study_design_scores_gemma":[0.0000055381934,0.000008680281,0.00002147872,0.0000056701438,0.0000033485449,0.0000043306745,0.0000014667545,0.99047416,0.00015877135,0.009223349,0.00009053061,0.0000027113024],"about_ca_topic_score_codex":0.0023021959,"about_ca_topic_score_gemma":0.0034686085,"teacher_disagreement_score":0.0026543164,"about_ca_system_score_codex":0.001380659,"about_ca_system_score_gemma":0.0010615304,"threshold_uncertainty_score":0.010017395},"labels":[],"label_agreement":null},{"id":"W4414183195","doi":"10.1038/s42256-025-01109-4","title":"Aligning generalization between humans and machines","year":2025,"lang":"en","type":"article","venue":"Nature Machine Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of British Columbia","funders":"Nederlandse Organisatie voor Wetenschappelijk Onderzoek","keywords":"Generalization; Abstraction; Cognition; Generative grammar; Key (lock); Human intelligence","score_opus":0.00935696220116273,"score_gpt":0.29798816362097796,"score_spread":0.28863120141981524,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414183195","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37319484,0.0007812868,0.5964955,0.003031842,0.00028823363,0.00013326523,0.00016835977,0.0020916366,0.023814926],"genre_scores_gemma":[0.9628608,0.00012280578,0.034376025,0.00025865083,0.00003514598,0.00003984274,0.00010184355,0.00009059846,0.0021142398],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99886763,0.0004960618,0.000045382865,0.00035767816,0.00014574938,0.00008757869],"domain_scores_gemma":[0.9938263,0.003223249,0.000421174,0.0017611623,0.00056437484,0.00020360934],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026251613,0.00040323584,0.00053831644,0.00040261974,0.00033718513,0.0010845585,0.00089151063,0.0011000182,0.0031268143],"category_scores_gemma":[0.014209396,0.00036019817,0.00042418408,0.00030917197,0.0014627157,0.0029970107,0.0019211225,0.0019338415,0.00068920257],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048490003,0.00038093646,0.017335184,0.00029132134,0.00035348555,0.00027244844,0.0012967053,0.37262818,0.02322149,0.09569973,0.0074920976,0.48054352],"study_design_scores_gemma":[0.000042094234,0.0002216321,0.0062032035,0.000034310342,0.000047303904,0.00012161846,0.00026673335,0.7551463,0.0050675375,0.2288994,0.003917234,0.00003254255],"about_ca_topic_score_codex":0.0024240909,"about_ca_topic_score_gemma":0.002269272,"teacher_disagreement_score":0.0031268143,"about_ca_system_score_codex":0.00079390424,"about_ca_system_score_gemma":0.0007799389,"threshold_uncertainty_score":0.013883352},"labels":[],"label_agreement":null},{"id":"W4414321910","doi":"10.1109/tai.2025.3610590","title":"Towards Sample-Efficiency and Generalization of Transfer and Inverse Reinforcement Learning: A Comprehensive Literature Review","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick; University of Windsor; Toronto Metropolitan University","funders":"","keywords":"Generalization; Inverse; Transfer (computing); Stability (learning theory); Calculus (dental)","score_opus":0.03715181315305159,"score_gpt":0.2995708591001717,"score_spread":0.2624190459471201,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414321910","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008214992,0.6322766,0.3478768,0.0025707977,0.00029192734,0.000078818084,0.0001301218,0.00024178415,0.008318222],"genre_scores_gemma":[0.26542985,0.57305926,0.15284963,0.0011683557,0.0028127327,0.00027851394,0.0005344722,0.00023028911,0.003636796],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99777156,0.00058785285,0.00025890663,0.00064360414,0.0006454478,0.000092480244],"domain_scores_gemma":[0.98248416,0.014509058,0.00048690106,0.000805641,0.00159855,0.00011563304],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005758259,0.0013862223,0.002589431,0.0018439534,0.00031793438,0.0026700215,0.0024073047,0.0017733265,0.0023332015],"category_scores_gemma":[0.024900198,0.0007147221,0.0011211423,0.0028469341,0.0016526041,0.00540449,0.001853953,0.002269119,0.000758059],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001264283,0.00015011443,0.001618788,0.0050015887,0.00026172033,0.00006209358,0.00013608791,0.04911924,0.0006780318,0.065176554,0.0040878444,0.87358147],"study_design_scores_gemma":[0.00009003968,0.00068744057,0.0065338057,0.004544764,0.0007240117,0.00092462194,0.00040565222,0.47715858,0.0038709966,0.42417642,0.080719985,0.00016364036],"about_ca_topic_score_codex":0.0027534293,"about_ca_topic_score_gemma":0.0017413774,"teacher_disagreement_score":0.005758259,"about_ca_system_score_codex":0.0013649052,"about_ca_system_score_gemma":0.0021199833,"threshold_uncertainty_score":0.030452907},"labels":[],"label_agreement":null},{"id":"W4414359474","doi":"10.24963/ijcai.2025/964","title":"DGL: Dynamic Global-Local Information Aggregation for Scalable VRP Generalization with Self-Improvement Learning","year":2025,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"BC Research (Canada)","funders":"","keywords":"Benchmark (surveying); Limiting; Scalability; Vehicle routing problem; Generalization; Context (archaeology); Node (physics)","score_opus":0.0033066275769797057,"score_gpt":0.22541397040296537,"score_spread":0.22210734282598568,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414359474","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02149999,0.00069311104,0.97185385,0.00055707584,0.000078482146,0.00012797248,0.00015976132,0.0026530211,0.0023768095],"genre_scores_gemma":[0.702213,0.00035265076,0.29173523,0.000950287,0.00012306437,0.0004555843,0.0007792201,0.00044679636,0.002944169],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99904054,0.00028985293,0.000058851823,0.0002655397,0.00022873822,0.00011639401],"domain_scores_gemma":[0.99741924,0.0014262536,0.0002636585,0.00038606717,0.0003464873,0.00015834777],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027288054,0.0017271232,0.001875043,0.0009851622,0.00059857866,0.00108401,0.0033567285,0.0016759519,0.002357022],"category_scores_gemma":[0.006473802,0.0008207979,0.0010130797,0.00093834085,0.0014084774,0.0021135763,0.0035022802,0.0028260374,0.00067824346],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000048259335,0.000108799315,0.0009465679,0.000083741674,0.0000510334,0.00006271232,0.000063593914,0.9285835,0.0007971076,0.003850064,0.0029669064,0.06243765],"study_design_scores_gemma":[0.00000901827,0.000020688141,0.000034011173,0.000004519578,0.0000034234,0.0000055065,0.000003856498,0.9977545,0.0001493348,0.0017943801,0.0002183548,0.0000025154686],"about_ca_topic_score_codex":0.0072158147,"about_ca_topic_score_gemma":0.008388922,"teacher_disagreement_score":0.0072158147,"about_ca_system_score_codex":0.001671424,"about_ca_system_score_gemma":0.0017448548,"threshold_uncertainty_score":0.014431477},"labels":[],"label_agreement":null},{"id":"W4414360545","doi":"10.24963/ijcai.2025/19","title":"Combining Deep Reinforcement Learning and Search with Generative Models for Game-Theoretic Opponent Modeling","year":2025,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Generative model; Scalability; Variety (cybernetics); Generative grammar; Class (philosophy); Tree (set theory); Adversary; Best response","score_opus":0.02852594522291188,"score_gpt":0.2784602814118119,"score_spread":0.2499343361889,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414360545","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02356947,0.00013510966,0.9735335,0.00031563724,0.000026601167,0.000034964174,0.00003330748,0.0005095072,0.0018419729],"genre_scores_gemma":[0.81976616,0.00013965397,0.1768413,0.00027933673,0.000041396874,0.00015493382,0.00013382625,0.00013549339,0.0025079024],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994288,0.00020104417,0.000025007384,0.00013159712,0.0001344419,0.000079169055],"domain_scores_gemma":[0.9979304,0.0013702959,0.00019282443,0.00020302091,0.00017243707,0.00013096632],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016065615,0.0010479775,0.0010429288,0.0006047497,0.00040800677,0.0010964208,0.0020534133,0.0012204342,0.002349821],"category_scores_gemma":[0.0057709725,0.00070100435,0.00072156347,0.00045934707,0.0014809665,0.0018830872,0.0017369847,0.0027261686,0.00041162007],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000032013457,0.00004709737,0.0007996274,0.000024244064,0.000034967496,0.00003876818,0.000052820295,0.959834,0.0006843552,0.016878767,0.00045095672,0.021122398],"study_design_scores_gemma":[0.0000025672077,0.0000051780594,0.000020133773,0.0000015863995,0.0000018072142,0.0000025893717,0.0000019242925,0.99486005,0.000101679085,0.004927122,0.00007392501,0.0000013782628],"about_ca_topic_score_codex":0.006198134,"about_ca_topic_score_gemma":0.00834666,"teacher_disagreement_score":0.006198134,"about_ca_system_score_codex":0.001558389,"about_ca_system_score_gemma":0.0013833322,"threshold_uncertainty_score":0.012324095},"labels":[],"label_agreement":null},{"id":"W4414746796","doi":"10.3390/sym17101632","title":"A Survey of Maximum Entropy-Based Inverse Reinforcement Learning: Methods and Applications","year":2025,"lang":"en","type":"article","venue":"Symmetry","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Ambiguity; Reinforcement learning; Benchmark (surveying); Principle of maximum entropy; Matching (statistics); Inverse","score_opus":0.02304166197990325,"score_gpt":0.32256258332432103,"score_spread":0.29952092134441777,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414746796","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018119041,0.060066577,0.9307208,0.00068421476,0.00016494829,0.000053651165,0.000057497735,0.00026881127,0.0061715585],"genre_scores_gemma":[0.30082273,0.15124902,0.53346324,0.0010119601,0.0016708445,0.00070311216,0.00050116424,0.0004908612,0.010087088],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988427,0.00041888093,0.000118555545,0.0002060722,0.00036434017,0.000049380313],"domain_scores_gemma":[0.9975305,0.0018004588,0.000157397,0.00012144986,0.00032086455,0.000069337184],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023826484,0.0016876961,0.0022709796,0.001486345,0.00039301394,0.0016331932,0.0016645241,0.0015187191,0.00281453],"category_scores_gemma":[0.0058838963,0.0008122942,0.0012351155,0.0018245365,0.0011905318,0.0016349372,0.0015750928,0.0022271643,0.000958637],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009449231,0.00015179442,0.0012105841,0.0021108743,0.00018600529,0.00010053073,0.00017624807,0.38805294,0.0022894987,0.08531741,0.0055485633,0.5147611],"study_design_scores_gemma":[0.000023939161,0.000140484,0.0004236824,0.00034709493,0.000049374758,0.00013753916,0.000031476975,0.9148386,0.0014896946,0.06123801,0.021217091,0.00006298894],"about_ca_topic_score_codex":0.0026586803,"about_ca_topic_score_gemma":0.001785785,"teacher_disagreement_score":0.00281453,"about_ca_system_score_codex":0.0012763645,"about_ca_system_score_gemma":0.0013932654,"threshold_uncertainty_score":0.012600839},"labels":[],"label_agreement":null},{"id":"W4415014469","doi":"10.1016/j.asoc.2025.114024","title":"Transformer-based dynamics model for sim-to-real reinforcement learning control of a quadrotor with limited experimental data","year":2025,"lang":"en","type":"article","venue":"Applied Soft Computing","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Institute for Information and Communications Technology Promotion; Cultural Heritage Administration; Information Technology Research Centre; National Research Institute of Cultural Heritage; Ministry of Science and ICT, South Korea","keywords":"Reinforcement learning; Transformer; Experimental data; Scheme (mathematics); Actuator; Controller (irrigation); Robot; Replicate","score_opus":0.022281285446842915,"score_gpt":0.28204203000989225,"score_spread":0.2597607445630493,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415014469","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017339235,0.00020628619,0.972844,0.00019973265,0.00008369084,0.000062998864,0.0001367371,0.00041682067,0.008710591],"genre_scores_gemma":[0.97678715,0.00015203418,0.01700346,0.00006555243,0.000017441538,0.0001469126,0.00010289532,0.000042718468,0.0056817797],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997937,0.00004415646,0.000012575159,0.0000549449,0.000065618726,0.000028844071],"domain_scores_gemma":[0.99971706,0.0000839838,0.000052869815,0.00003000023,0.00009540629,0.000020569167],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004928659,0.00062279805,0.00085073424,0.0003375484,0.00031615223,0.00068698416,0.0011090286,0.0009515705,0.0046569663],"category_scores_gemma":[0.0008077006,0.0003133448,0.0005303773,0.0002768668,0.0006942489,0.0007186918,0.0010818631,0.0008360759,0.0007154847],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009023334,0.000025681406,0.00022797871,0.00009764258,0.000022813974,0.000092222384,0.000047131973,0.97203976,0.0042993114,0.009989512,0.00050898094,0.0125587145],"study_design_scores_gemma":[0.000005968221,0.000017575687,0.000053080257,0.0000026936102,0.00000329511,0.000009186499,0.000002993338,0.9985312,0.00026761147,0.00091756147,0.00018620017,0.0000026139178],"about_ca_topic_score_codex":0.0065729744,"about_ca_topic_score_gemma":0.004738595,"teacher_disagreement_score":0.0065729744,"about_ca_system_score_codex":0.0005190545,"about_ca_system_score_gemma":0.0007530169,"threshold_uncertainty_score":0.015579104},"labels":[],"label_agreement":null},{"id":"W4415048913","doi":"10.1109/tpami.2025.3619883","title":"Release the Potential of Memory Buffer in Continual Learning: A Dynamic System Perspective","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Pattern Analysis and Machine Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Forgetting; Overfitting; Memory model; Memory management; Dynamic random-access memory; Transformation (genetics); Adaptive memory; Perspective (graphical)","score_opus":0.007327819433915642,"score_gpt":0.25508305915173934,"score_spread":0.2477552397178237,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415048913","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035739124,0.00048259727,0.9603258,0.0005269585,0.000052225143,0.000048110312,0.00006351701,0.00050262257,0.0022590973],"genre_scores_gemma":[0.9495411,0.00038665137,0.04567406,0.00027579072,0.00006030333,0.00013610291,0.000085236104,0.000112297595,0.0037283662],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994716,0.0001158587,0.000035379686,0.0001632021,0.00012604691,0.0000879638],"domain_scores_gemma":[0.99752957,0.0013620411,0.00028407917,0.00044197333,0.00025182235,0.00013043729],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016280947,0.0009875355,0.0009293947,0.00040205172,0.00037655656,0.0011669686,0.0022015697,0.001114648,0.002710241],"category_scores_gemma":[0.0071250033,0.00049934414,0.0006349295,0.0003750242,0.0017163156,0.0033357877,0.0027847758,0.0025057863,0.00040571476],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019180414,0.000095228424,0.0011354091,0.00015338916,0.00007135824,0.00016951405,0.00019783793,0.8762462,0.006265976,0.047523387,0.001123852,0.0668261],"study_design_scores_gemma":[0.0000096680005,0.0000727256,0.000090600006,0.000011892808,0.0000122800775,0.000043549073,0.0000133491585,0.98018056,0.0018320372,0.017245697,0.0004774476,0.00001012095],"about_ca_topic_score_codex":0.0018548877,"about_ca_topic_score_gemma":0.0016082474,"teacher_disagreement_score":0.002710241,"about_ca_system_score_codex":0.00087079254,"about_ca_system_score_gemma":0.0011179514,"threshold_uncertainty_score":0.009066701},"labels":[],"label_agreement":null},{"id":"W4415112009","doi":"10.48550/arxiv.2506.12622","title":"DR-SAC: Distributionally Robust Soft Actor-Critic for Reinforcement Learning under Uncertainty","year":2025,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Division of Electrical, Communications and Cyber Systems; Office of Naval Research; York University; National Science Foundation","keywords":"Robustness (evolution); Reinforcement learning; Robust control; Benchmark (surveying); Entropy (arrow of time); Convergence (economics); Robust optimization; Rate of convergence","score_opus":0.07787399934876577,"score_gpt":0.21875883121698483,"score_spread":0.14088483186821907,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415112009","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0055740033,0.00018993697,0.99080324,0.00017711654,0.000058225756,0.000043354594,0.000041559088,0.0012238414,0.0018886866],"genre_scores_gemma":[0.719243,0.00028137641,0.27309754,0.00052782893,0.0001087413,0.000287207,0.00036410216,0.0006125569,0.0054777],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987936,0.00041689206,0.000070217946,0.0002817394,0.00031158538,0.00012592238],"domain_scores_gemma":[0.997072,0.0018258829,0.00027430392,0.00026782937,0.00038020985,0.0001797175],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022176607,0.0015488724,0.0015511496,0.0005233198,0.00046179752,0.0011749992,0.0019449606,0.0015102675,0.003493189],"category_scores_gemma":[0.008306174,0.00066290307,0.0007832551,0.0004363939,0.0015639293,0.0011903635,0.0021608223,0.0031705345,0.00097050617],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007551834,0.0000384837,0.00042308882,0.00007282795,0.000048288246,0.000060887178,0.000045113506,0.94546884,0.0012445167,0.011578149,0.0018042969,0.03913993],"study_design_scores_gemma":[0.000006859689,0.000011774585,0.000019046172,0.0000047851736,0.0000027724268,0.000007359007,0.0000019600536,0.99607134,0.00025479434,0.0033965264,0.00021969822,0.0000030829708],"about_ca_topic_score_codex":0.004519799,"about_ca_topic_score_gemma":0.0051110904,"teacher_disagreement_score":0.004519799,"about_ca_system_score_codex":0.0011838045,"about_ca_system_score_gemma":0.002388311,"threshold_uncertainty_score":0.011728287},"labels":[],"label_agreement":null},{"id":"W4415318339","doi":"10.48550/arxiv.2510.07257","title":"Test-Time Graph Search for Goal-Conditioned Reinforcement Learning","year":2025,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Government of Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Graph; Metric (unit); Bounding overwatch; Base (topology); Train; Learning to rank; Sequence (biology)","score_opus":0.048363397902601656,"score_gpt":0.2102282857347292,"score_spread":0.16186488783212755,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415318339","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.058545154,0.0006710974,0.9235502,0.0006412203,0.00014420839,0.00018374133,0.000558823,0.010246339,0.0054592574],"genre_scores_gemma":[0.74542266,0.00019200816,0.24808368,0.0004277494,0.0000476582,0.0003245254,0.001482309,0.00093818526,0.0030813057],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993229,0.00023257539,0.000032518896,0.00021144669,0.000108393135,0.00009211707],"domain_scores_gemma":[0.9975979,0.0014567338,0.00014356327,0.000430759,0.00020230378,0.00016885117],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012505042,0.0015383909,0.0010177789,0.00057146983,0.00044039465,0.00080695783,0.002323838,0.001351081,0.005393633],"category_scores_gemma":[0.0070163156,0.00050565606,0.00059545616,0.00053052977,0.0012042674,0.0015397399,0.0016822217,0.0024780014,0.0012521072],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030763904,0.0002265424,0.0021310458,0.00022006487,0.000065570566,0.00011771283,0.00008624499,0.8158912,0.00272208,0.009475751,0.008851841,0.15990427],"study_design_scores_gemma":[0.000033053104,0.000043921038,0.00010993362,0.000010939986,0.000006446257,0.0000125119695,0.0000109787,0.9886874,0.0007958654,0.009593284,0.00069072854,0.0000049660694],"about_ca_topic_score_codex":0.006143304,"about_ca_topic_score_gemma":0.0096344305,"teacher_disagreement_score":0.006143304,"about_ca_system_score_codex":0.0012008608,"about_ca_system_score_gemma":0.0022491727,"threshold_uncertainty_score":0.018043458},"labels":[],"label_agreement":null},{"id":"W4415428256","doi":"10.3233/faia251090","title":"Skill-Enhanced Reinforcement Learning Acceleration from Heterogeneous Demonstrations","year":2025,"lang":"","type":"book-chapter","venue":"Frontiers in artificial intelligence and applications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Institute for Advanced Research; Carleton University","funders":"","keywords":"Reinforcement learning; Leverage (statistics); Downstream (manufacturing); Temporal difference learning; Acceleration; Supervised learning; Online machine learning; Reinforcement","score_opus":0.03288550673780755,"score_gpt":0.2740642108823426,"score_spread":0.24117870414453507,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415428256","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014655688,0.00025505014,0.97950244,0.00011515037,0.000068676985,0.00005509117,0.00004534304,0.001284559,0.004018047],"genre_scores_gemma":[0.67946553,0.00036338723,0.30807543,0.0001955533,0.000080294565,0.00022885423,0.00028297992,0.00029502137,0.011013021],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997317,0.000055937293,0.000013060455,0.00007607094,0.00008271822,0.00004044384],"domain_scores_gemma":[0.9991799,0.00044695352,0.00007805954,0.00012205718,0.0001155311,0.000057511184],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007158076,0.00091442943,0.00071383687,0.00022138092,0.00022039002,0.0005124105,0.0011907105,0.00062136527,0.0049071005],"category_scores_gemma":[0.0023206947,0.0003825269,0.00041339587,0.00020765806,0.0005097781,0.000823068,0.0015106688,0.001868487,0.0011494817],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001525866,0.00010798095,0.00061206945,0.00013071229,0.000032294098,0.00010918234,0.000069714144,0.75075656,0.014373997,0.01091862,0.0038373899,0.21889888],"study_design_scores_gemma":[0.000007392037,0.00003307871,0.000073781805,0.0000049773394,0.0000028168217,0.000015340485,0.000002231958,0.9955584,0.001537036,0.0019841462,0.0007776156,0.0000031525342],"about_ca_topic_score_codex":0.0020374302,"about_ca_topic_score_gemma":0.0024196312,"teacher_disagreement_score":0.0049071005,"about_ca_system_score_codex":0.00043257562,"about_ca_system_score_gemma":0.0008388574,"threshold_uncertainty_score":0.016415894},"labels":[],"label_agreement":null},{"id":"W4415428281","doi":"10.3233/faia251077","title":"DmC: Nearest Neighbor Guidance Diffusion Model for Offline Cross-Domain Reinforcement Learning","year":2025,"lang":"","type":"book-chapter","venue":"Frontiers in artificial intelligence and applications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Overfitting; Reinforcement learning; Domain (mathematical analysis); Key (lock); Artificial neural network; Sample (material); k-nearest neighbors algorithm","score_opus":0.04380986476356294,"score_gpt":0.3078464961854975,"score_spread":0.26403663142193456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415428281","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017683271,0.00049769,0.97800076,0.0003425663,0.00006365251,0.000091935304,0.00008827195,0.0005885178,0.0026432371],"genre_scores_gemma":[0.8820152,0.00027247088,0.11155831,0.00045371163,0.000051945324,0.00036461957,0.00026617103,0.00013752392,0.004880091],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99893373,0.00031999455,0.00005448154,0.00030716602,0.0002523221,0.00013221947],"domain_scores_gemma":[0.9968406,0.0017713598,0.0003429786,0.00026565278,0.0005733374,0.0002060899],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024498655,0.0012382222,0.0018277951,0.00063787174,0.0005225024,0.0012580174,0.0025742976,0.0016806505,0.002661891],"category_scores_gemma":[0.007900404,0.00058184343,0.0006183102,0.00059647346,0.001451988,0.0016954957,0.0020264853,0.002910152,0.0005755633],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010210066,0.000089684385,0.00085409114,0.00006776677,0.000037669677,0.00007337089,0.0000740647,0.9516385,0.0009806553,0.008955542,0.001494117,0.0356323],"study_design_scores_gemma":[0.0000053816434,0.000013977878,0.000039342787,0.000003516309,0.0000021471042,0.000005534437,0.0000026459068,0.99825364,0.00012933127,0.0013664227,0.00017449277,0.0000035735773],"about_ca_topic_score_codex":0.010035035,"about_ca_topic_score_gemma":0.008061335,"teacher_disagreement_score":0.010035035,"about_ca_system_score_codex":0.0015537211,"about_ca_system_score_gemma":0.0020285163,"threshold_uncertainty_score":0.01995325},"labels":[],"label_agreement":null},{"id":"W4415588118","doi":"10.1021/acs.iecr.5c01979","title":"Hierarchical Reinforcement Learning with Dynamic Meta Agent for Adaptive Cut Selection in Integer Programming with Applications to Sensor Network Design","year":2025,"lang":"en","type":"article","venue":"Industrial & Engineering Chemistry Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Science and Engineering Research Board; Indian Institute of Technology Delhi; Cornell University","keywords":"Reinforcement learning; Integer programming; Selection (genetic algorithm); Scalability; Cutting-plane method; Linear programming; Integer (computer science); Wireless sensor network","score_opus":0.08648733691679072,"score_gpt":0.3346498647927557,"score_spread":0.248162527875965,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415588118","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020981295,0.00025251092,0.9760926,0.00019414432,0.000027164573,0.000056765377,0.000017751634,0.00035923967,0.002018469],"genre_scores_gemma":[0.8162085,0.00017031598,0.18154483,0.0001917425,0.000037140293,0.00020153282,0.0000521948,0.000065076405,0.0015286308],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994661,0.0002160014,0.000021530903,0.00009921875,0.00011498791,0.00008211579],"domain_scores_gemma":[0.9985682,0.0009294371,0.00019511282,0.000082650724,0.00013606956,0.00008856479],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001362954,0.000864665,0.0008175222,0.0005107543,0.00033200264,0.0006842209,0.0011772313,0.0007200694,0.0018493104],"category_scores_gemma":[0.0031040967,0.00040852342,0.00049376994,0.0003765431,0.0009828582,0.00077891385,0.0010949157,0.0012883107,0.00024389352],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003428168,0.00005007444,0.00041970058,0.000035396723,0.000021086598,0.000031193144,0.000032516105,0.97182816,0.0010323412,0.0056265313,0.00032148702,0.020567317],"study_design_scores_gemma":[0.000006149337,0.00001752612,0.000023470036,0.0000022345525,0.000002314254,0.0000033762267,0.0000027182973,0.9981395,0.00014959414,0.0015298892,0.000121953344,0.0000013933433],"about_ca_topic_score_codex":0.0030293756,"about_ca_topic_score_gemma":0.0036453581,"teacher_disagreement_score":0.0030293756,"about_ca_system_score_codex":0.0010127086,"about_ca_system_score_gemma":0.0012398119,"threshold_uncertainty_score":0.0073477626},"labels":[],"label_agreement":null},{"id":"W4415688799","doi":"10.1007/978-3-032-04339-9_15","title":"Enhancing Off-Policy Method SAC with KAN for Continuous Reinforcement Learning","year":2025,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Regina","funders":"","keywords":"Reinforcement learning; Embedding; Architecture; Reinforcement; Artificial neural network","score_opus":0.02445174135414655,"score_gpt":0.3154284189345152,"score_spread":0.2909766775803686,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415688799","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007394944,0.0001790956,0.98589677,0.000080858386,0.00014460417,0.000040186194,0.000017211494,0.0008278581,0.0054184515],"genre_scores_gemma":[0.58914655,0.00024189282,0.3960967,0.00022686082,0.00014102271,0.00017816473,0.00007991992,0.00039715084,0.013491759],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99949336,0.00014245263,0.000024006693,0.00009898725,0.00018369858,0.00005751118],"domain_scores_gemma":[0.9989667,0.00048448038,0.000063106076,0.00016733688,0.00023326468,0.00008513362],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010695512,0.0007107896,0.0008546647,0.00046255218,0.00039830743,0.00086501695,0.0012069581,0.00095676456,0.0075446875],"category_scores_gemma":[0.0025545135,0.0002698355,0.00047013868,0.00031204856,0.0008241628,0.0010551754,0.0016046995,0.0018730216,0.0014625695],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005029992,0.00032594657,0.0007405972,0.00028579973,0.0000768232,0.00013647991,0.00013286257,0.49090207,0.016365988,0.058423787,0.005674529,0.42643222],"study_design_scores_gemma":[0.000015766573,0.00007306655,0.00006942573,0.0000069659345,0.0000071103404,0.0000361405,0.0000058060414,0.9907699,0.0013642899,0.0060743224,0.0015711912,0.0000060460857],"about_ca_topic_score_codex":0.0017002311,"about_ca_topic_score_gemma":0.0019232166,"teacher_disagreement_score":0.0075446875,"about_ca_system_score_codex":0.0006032768,"about_ca_system_score_gemma":0.0012756882,"threshold_uncertainty_score":0.025239527},"labels":[],"label_agreement":null},{"id":"W4415935956","doi":"10.1016/j.cobeha.2025.101611","title":"Linking homeostasis to reinforcement learning: internal state control of motivated behavior","year":2025,"lang":"en","type":"article","venue":"Current Opinion in Behavioral Sciences","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Japan Society for the Promotion of Science; Conseil National de la Recherche Scientifique; Agence Nationale de la Recherche; Institut National de la Santé et de la Recherche Médicale; Engineers Nova Scotia","keywords":"Reinforcement learning; Adaptive behavior; Reinforcement; Cognition; Internal model; Control (management); Embodied cognition","score_opus":0.06681357893934883,"score_gpt":0.379216424393706,"score_spread":0.31240284545435715,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415935956","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.056723256,0.0015131272,0.92062694,0.0026938985,0.0001846852,0.000034587865,0.00009687344,0.00032349493,0.017803125],"genre_scores_gemma":[0.95018923,0.00090518076,0.045426987,0.00041523573,0.000116071715,0.00008115702,0.00005471014,0.00006253698,0.0027487935],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9997614,0.00008104602,0.000011583853,0.00007036372,0.000044459088,0.000031188476],"domain_scores_gemma":[0.9995158,0.00017548553,0.00011995353,0.00007310973,0.000058959835,0.000056638917],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00053045846,0.00043696858,0.00036162246,0.00023614684,0.00028776212,0.0013223378,0.0008498577,0.00079932687,0.0021034009],"category_scores_gemma":[0.001748856,0.00018527909,0.00053086434,0.00016336856,0.002222696,0.0017413717,0.0009813313,0.001317018,0.0003259948],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009900025,0.00009079593,0.004188981,0.00018406643,0.000121174504,0.00021611685,0.0003528571,0.2140057,0.017474804,0.70068866,0.0023389931,0.060238976],"study_design_scores_gemma":[0.000015272708,0.00008529171,0.0015103075,0.00003601418,0.0000209275,0.000091772505,0.000049094742,0.44487256,0.0018713103,0.548074,0.0033368366,0.000036539022],"about_ca_topic_score_codex":0.0011254154,"about_ca_topic_score_gemma":0.00054645434,"teacher_disagreement_score":0.0021034009,"about_ca_system_score_codex":0.0006350989,"about_ca_system_score_gemma":0.00051719527,"threshold_uncertainty_score":0.007036507},"labels":[],"label_agreement":null},{"id":"W4416036933","doi":"10.18653/v1/2025.emnlp-main.125","title":"REARANK: Reasoning Re-ranking Agent via Reinforcement Learning","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Canadian Institute for Advanced Research; Nvidia","keywords":"Reinforcement learning; Control (management); Action (physics); Feature (linguistics); Stability (learning theory)","score_opus":0.016838529688993045,"score_gpt":0.26877404190024884,"score_spread":0.2519355122112558,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416036933","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027021587,0.00027316532,0.93629307,0.00050617586,0.00035689303,0.00028294988,0.00042285278,0.02640564,0.008437721],"genre_scores_gemma":[0.4433547,0.00015521303,0.53256434,0.0004390141,0.00007345252,0.00027836856,0.0008937398,0.0010916187,0.021149551],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99950993,0.00011351297,0.000028854865,0.0001163692,0.00016627996,0.0000650154],"domain_scores_gemma":[0.99933547,0.00024099823,0.000040276034,0.00015122631,0.00015796715,0.00007403659],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009782433,0.0010693315,0.0010674242,0.0005415871,0.0005828331,0.00094903534,0.0024726843,0.0014011833,0.012947916],"category_scores_gemma":[0.0029536493,0.000534351,0.0006628607,0.00024625866,0.00062629924,0.0013699634,0.0017339662,0.0020144247,0.0029472949],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008586968,0.000780578,0.0017718258,0.0004115986,0.00022760796,0.000469561,0.00015075061,0.42147085,0.0142422635,0.028719222,0.040902376,0.4899948],"study_design_scores_gemma":[0.00009166304,0.000090183865,0.00010222085,0.000010399405,0.000026223259,0.00004780876,0.000013055662,0.98244494,0.0040624407,0.009352492,0.0037417803,0.000016739343],"about_ca_topic_score_codex":0.0058241216,"about_ca_topic_score_gemma":0.010935037,"teacher_disagreement_score":0.012947916,"about_ca_system_score_codex":0.00065226643,"about_ca_system_score_gemma":0.001493411,"threshold_uncertainty_score":0.043315113},"labels":[],"label_agreement":null},{"id":"W4416070655","doi":"10.48550/arxiv.2505.20290","title":"EgoZero: Robot Learning from Smart Glasses","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Office of Naval Research; York University; National Science Foundation","keywords":"Robot; Robot learning; Human–robot interaction; Scalability; Robotics; Programming by demonstration; Social robot","score_opus":0.05367968038562613,"score_gpt":0.2787033435342162,"score_spread":0.2250236631485901,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416070655","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.085432395,0.0005027525,0.8542096,0.00041912266,0.00014627639,0.0002341497,0.0011455286,0.05298705,0.004923189],"genre_scores_gemma":[0.7193755,0.00025685527,0.2681459,0.00035395913,0.000031838663,0.00043100162,0.0028988617,0.0015200004,0.0069861296],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997323,0.000050395738,0.000011250451,0.00010878111,0.000061808714,0.000035430316],"domain_scores_gemma":[0.9996395,0.00013635644,0.000032783642,0.00012000752,0.000037765065,0.00003353511],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00053019286,0.000709359,0.00053057424,0.00026057966,0.0002634334,0.00043432444,0.0016191269,0.00071359065,0.0043755556],"category_scores_gemma":[0.0023048948,0.00046770863,0.00049276795,0.00021059412,0.000855654,0.0010004529,0.002124119,0.0010019639,0.0013342204],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007536606,0.00033694366,0.0042015193,0.00041107682,0.00017674547,0.00025160084,0.00030155166,0.4664026,0.045334868,0.011458627,0.025120964,0.4452498],"study_design_scores_gemma":[0.00004282951,0.0001410267,0.00079572527,0.000020063797,0.000011855671,0.000045478395,0.000028177325,0.9730886,0.011313344,0.010032377,0.0044595785,0.000020862788],"about_ca_topic_score_codex":0.006256568,"about_ca_topic_score_gemma":0.008089122,"teacher_disagreement_score":0.006256568,"about_ca_system_score_codex":0.0005231268,"about_ca_system_score_gemma":0.0008450002,"threshold_uncertainty_score":0.014637649},"labels":[],"label_agreement":null},{"id":"W4416125924","doi":"10.54254/2755-2721/2026.tj29489","title":"On-Policy Vs. Off-Policy Reinforcement Learning in ConnectX: Seat-Stratified Performance and the Role of Action Masking","year":2025,"lang":"","type":"article","venue":"Applied and Computational Engineering","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Pooling; Robustness (evolution); Generalization; Masking (illustration); Reinforcement learning; Reinforcement; Lever; Set (abstract data type)","score_opus":0.005874809224640797,"score_gpt":0.2240839369894798,"score_spread":0.218209127764839,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416125924","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8116808,0.0034294794,0.16169363,0.0014713464,0.00044467882,0.00053851726,0.00041720286,0.0057301885,0.014594057],"genre_scores_gemma":[0.9770817,0.00018209132,0.019553246,0.000363614,0.000033203854,0.0001692114,0.00028527848,0.00014451225,0.00218714],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984433,0.0005934987,0.000075852004,0.00035863637,0.00025748403,0.00027130067],"domain_scores_gemma":[0.9947015,0.0032764017,0.00042168083,0.00063566724,0.00040598947,0.00055877777],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052040406,0.0016407798,0.0012891521,0.0004161379,0.00042048038,0.001036377,0.0017484066,0.0019420664,0.0032744482],"category_scores_gemma":[0.0145564135,0.00038943175,0.0004908455,0.00019908856,0.0015794089,0.00188698,0.00179073,0.003095128,0.0007815962],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0062381923,0.0015624235,0.012918712,0.0005536396,0.00029155178,0.00016944595,0.00020019858,0.8075521,0.009735447,0.0058474327,0.0051567648,0.14977404],"study_design_scores_gemma":[0.00038256004,0.0028338225,0.0026766395,0.00008542307,0.00006876909,0.000073039744,0.00007999157,0.9808255,0.0066702035,0.0048766662,0.0013802337,0.000047223395],"about_ca_topic_score_codex":0.0040902467,"about_ca_topic_score_gemma":0.0044500753,"teacher_disagreement_score":0.0052040406,"about_ca_system_score_codex":0.0010949684,"about_ca_system_score_gemma":0.0025284512,"threshold_uncertainty_score":0.027521908},"labels":[],"label_agreement":null},{"id":"W4416177363","doi":"10.48550/arxiv.2508.06836","title":"Multi-level Advantage Credit Assignment for Cooperative Multi-Agent Reinforcement Learning","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada","keywords":"Counterfactual thinking; Reinforcement learning; Key (lock); Diversity (politics); Function (biology)","score_opus":0.11935644116154431,"score_gpt":0.3388184746199345,"score_spread":0.21946203345839016,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416177363","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022045573,0.00012742875,0.9758839,0.000260638,0.000026633596,0.00005118024,0.000025023932,0.00028050275,0.0012990921],"genre_scores_gemma":[0.8914877,0.00007377663,0.10678569,0.00013474353,0.000039352486,0.00012853541,0.00004146296,0.000052133775,0.0012565727],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986332,0.00059803063,0.00006501049,0.00025340836,0.000305298,0.00014499073],"domain_scores_gemma":[0.9956975,0.002675413,0.000481415,0.00040440858,0.000402314,0.00033894167],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030511112,0.0009003673,0.0011678437,0.0006781611,0.000508724,0.0009999932,0.0018635526,0.0011471518,0.0020737536],"category_scores_gemma":[0.008932659,0.00047000716,0.0005720793,0.00053440116,0.0016651779,0.0015062769,0.0019323464,0.0020909728,0.000230403],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009587867,0.0000903922,0.0018059039,0.00006352387,0.00005923318,0.00007730371,0.00011868896,0.92156583,0.0011485816,0.035904214,0.0008150208,0.038255453],"study_design_scores_gemma":[0.0000059577337,0.000011550305,0.000058555008,0.000002338736,0.0000033716199,0.0000048413403,0.0000029057035,0.98834455,0.00011454088,0.011339413,0.000109257526,0.000002751802],"about_ca_topic_score_codex":0.003272984,"about_ca_topic_score_gemma":0.0034291102,"teacher_disagreement_score":0.003272984,"about_ca_system_score_codex":0.0018049285,"about_ca_system_score_gemma":0.0016069112,"threshold_uncertainty_score":0.01613605},"labels":[],"label_agreement":null},{"id":"W4416184234","doi":"10.1109/acsos-c66519.2025.00053","title":"Improving Adaptability in Agents Through Popperian Expectations","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Adaptability; Balance (ability); Reinforcement learning; Control (management); Flexibility (engineering); Behaviour change; Variety (cybernetics)","score_opus":0.030693769037560895,"score_gpt":0.3049711465294727,"score_spread":0.27427737749191183,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416184234","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18543677,0.00017681149,0.7891687,0.0014828498,0.0000501715,0.000094159,0.000051167874,0.0007734015,0.022765921],"genre_scores_gemma":[0.94325775,0.0001265762,0.05414417,0.00019187569,0.000011822927,0.00006984438,0.000047661626,0.00008069956,0.0020696388],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990145,0.0004524707,0.000047337486,0.00015571268,0.00023012515,0.00009996337],"domain_scores_gemma":[0.9965777,0.0016768863,0.00048106367,0.0005989504,0.00034848254,0.0003170362],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020321172,0.0006329586,0.00039251297,0.00028283894,0.00047131992,0.0017673242,0.0010690306,0.00082470174,0.0018528812],"category_scores_gemma":[0.010800638,0.0003934107,0.00042236754,0.00019314447,0.0017086855,0.0035636534,0.0028046812,0.0015527287,0.00057205756],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031854404,0.00024591878,0.010360646,0.00020562652,0.000121598925,0.00054348086,0.002892208,0.62091863,0.022647526,0.22856312,0.0021514392,0.11103119],"study_design_scores_gemma":[0.000051568924,0.00018567585,0.0011424924,0.000037367612,0.0000391824,0.000121268204,0.00037743044,0.80372554,0.007078242,0.18167518,0.0055087227,0.000057359517],"about_ca_topic_score_codex":0.0011039906,"about_ca_topic_score_gemma":0.0011717627,"teacher_disagreement_score":0.0020321172,"about_ca_system_score_codex":0.0007641052,"about_ca_system_score_gemma":0.00090796873,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W4416198962","doi":"10.3390/sym17111951","title":"Dynamic Heterogeneous Multi-Agent Inverse Reinforcement Learning Based on Graph Attention Mean Field","year":2025,"lang":"en","type":"article","venue":"Symmetry","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Reinforcement learning; Ambiguity; Graph; Entropy (arrow of time); Principle of maximum entropy; Network topology; Inverse; Adversarial system","score_opus":0.013227040529980387,"score_gpt":0.2619231187260152,"score_spread":0.24869607819603481,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416198962","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024219288,0.00022178846,0.97347736,0.0002195797,0.000035387908,0.000032628366,0.000029103,0.00031848362,0.0014463357],"genre_scores_gemma":[0.93383646,0.00014503415,0.063113615,0.00019530872,0.000035950306,0.00010711735,0.00009146291,0.000052252144,0.0024227842],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994993,0.00015018905,0.0000198212,0.00013992337,0.000108465014,0.00008238333],"domain_scores_gemma":[0.9988501,0.00065966166,0.00014556205,0.00007885744,0.00017593653,0.00008983911],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010044699,0.0008883719,0.0011989498,0.00061190256,0.0003888723,0.0007327453,0.001605466,0.0009992884,0.0012537546],"category_scores_gemma":[0.0033353881,0.00043095293,0.00067912054,0.0004085721,0.0009954703,0.0010876848,0.0010943094,0.0013337389,0.00017939963],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003771273,0.0000414969,0.0008504427,0.000032804906,0.000039397397,0.000069493166,0.00004524677,0.9600262,0.0011455154,0.009301245,0.0006247619,0.027785702],"study_design_scores_gemma":[0.0000032092973,0.000007612154,0.000042281095,0.0000013044875,0.0000026277064,0.000004922995,0.00000165345,0.99763906,0.00009470954,0.0021320982,0.00006845041,0.0000021943902],"about_ca_topic_score_codex":0.0091667455,"about_ca_topic_score_gemma":0.005183702,"teacher_disagreement_score":0.0091667455,"about_ca_system_score_codex":0.0012282459,"about_ca_system_score_gemma":0.0011624868,"threshold_uncertainty_score":0.018226802},"labels":[],"label_agreement":null},{"id":"W4416251251","doi":"10.1109/ijcnn64981.2025.11228644","title":"Leveraging World Model Disentanglement in Value-Based Multi-Agent Reinforcement Learning","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Scalability; Sample (material); Function (biology); Graph; Range (aeronautics); Sample complexity; Sampling (signal processing)","score_opus":0.040929635774792084,"score_gpt":0.2967428688077293,"score_spread":0.25581323303293724,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416251251","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014847937,0.00016041928,0.98382336,0.00013665964,0.000019579575,0.000029992094,0.000024422427,0.00025164793,0.00070606667],"genre_scores_gemma":[0.86307853,0.00015972006,0.13461348,0.00021396157,0.000040327024,0.0001636723,0.00016080998,0.000108513625,0.0014610009],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993068,0.00027392793,0.00003115242,0.00015122717,0.00015514315,0.00008171971],"domain_scores_gemma":[0.99821585,0.0010920304,0.00020930413,0.00016081349,0.00018397995,0.0001379553],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015747208,0.0012027817,0.001788891,0.0005777486,0.00039600683,0.0008981135,0.0019780472,0.001265683,0.0014342073],"category_scores_gemma":[0.004221206,0.00082166534,0.0007037628,0.00050640467,0.0013250912,0.0017330024,0.0018120188,0.0020724898,0.00032222876],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004514104,0.00005729828,0.00078857335,0.00004756014,0.000046624325,0.00005592145,0.000049836235,0.95868766,0.0011242745,0.008850578,0.00044395978,0.029802654],"study_design_scores_gemma":[0.000004028913,0.000011218899,0.000024881909,0.0000023701568,0.000002522884,0.0000042926836,0.0000017969265,0.996727,0.000114185525,0.0030353377,0.00007020317,0.0000020862533],"about_ca_topic_score_codex":0.0040541664,"about_ca_topic_score_gemma":0.004478658,"teacher_disagreement_score":0.0040541664,"about_ca_system_score_codex":0.000948779,"about_ca_system_score_gemma":0.0012790089,"threshold_uncertainty_score":0.008328021},"labels":[],"label_agreement":null},{"id":"W4416251343","doi":"10.1109/ijcnn64981.2025.11227725","title":"Navigation With QPHIL: Quantizing Planner for Hierarchical Implicit Q-Learning","year":2025,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ubisoft (Canada)","funders":"","keywords":"Reinforcement learning; Trajectory; Planner; Image stitching; Motion planning; Path (computing); State space; Range (aeronautics); Function (biology)","score_opus":0.011123300682575932,"score_gpt":0.2746747569265827,"score_spread":0.2635514562440068,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416251343","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0045498465,0.0000901527,0.99327576,0.000054772758,0.000019486813,0.00003873347,0.000035551082,0.00087443396,0.0010613487],"genre_scores_gemma":[0.654678,0.0001751845,0.34103012,0.00018190438,0.000038677787,0.00025038322,0.00021728418,0.00022500314,0.0032034644],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99965465,0.00007077115,0.000021733897,0.00008996294,0.00011542528,0.00004748365],"domain_scores_gemma":[0.9991954,0.00044783214,0.00006486231,0.00012831793,0.000105577215,0.000057950107],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007936036,0.0007112768,0.00084417616,0.00029566436,0.00028598434,0.0006638096,0.0016254414,0.00081637973,0.0049441345],"category_scores_gemma":[0.002516919,0.00041389314,0.00040631002,0.00033457766,0.0009886274,0.0009808681,0.0017507109,0.0017423317,0.0007827436],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012895453,0.00007083249,0.00047327203,0.00014732056,0.00002387761,0.00008507514,0.00014491918,0.81331086,0.0036148797,0.024481315,0.0019728744,0.1555458],"study_design_scores_gemma":[0.000014777241,0.00002172505,0.000026186954,0.000004581501,0.0000026685764,0.000007236454,0.0000041183853,0.9950531,0.00038751945,0.00409005,0.00038530782,0.0000026368402],"about_ca_topic_score_codex":0.0051820865,"about_ca_topic_score_gemma":0.0065027415,"teacher_disagreement_score":0.0051820865,"about_ca_system_score_codex":0.0006565709,"about_ca_system_score_gemma":0.0017523025,"threshold_uncertainty_score":0.016539812},"labels":[],"label_agreement":null},{"id":"W4416386275","doi":"10.48550/arxiv.2510.08526","title":"Convergence Theorems for Entropy-Regularized and Distributional Reinforcement Learning","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Defense Advanced Research Projects Agency; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Decoupling (probability); Entropy (arrow of time); Regularization (linguistics); Convergence (economics); Principle of maximum entropy","score_opus":0.028096724574999912,"score_gpt":0.27354029106517214,"score_spread":0.24544356649017224,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416386275","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0059059327,0.000364618,0.9877429,0.00057732995,0.000054845765,0.00005895657,0.00007419924,0.00023340092,0.004987789],"genre_scores_gemma":[0.5720179,0.0017197995,0.40897205,0.0010038033,0.0003749276,0.001230964,0.0005196867,0.0009863172,0.013174539],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9973247,0.0012465107,0.00012562511,0.0004386579,0.0006430481,0.00022131192],"domain_scores_gemma":[0.981556,0.014331961,0.00080361014,0.0010050421,0.001756099,0.000547335],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009642614,0.001796864,0.0015645027,0.0020233123,0.0009845879,0.002119945,0.0021675704,0.0021356214,0.0054944498],"category_scores_gemma":[0.039924134,0.000745546,0.0018154045,0.00097338425,0.0044415747,0.0038103668,0.003990263,0.0053855916,0.0010418266],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009906062,0.00007972341,0.001033173,0.00020667618,0.00007820185,0.00008291947,0.00023052689,0.30237544,0.0012831955,0.6654957,0.0022440427,0.026791377],"study_design_scores_gemma":[0.000017866567,0.00003356992,0.00016302943,0.00004726403,0.000011116056,0.000027244794,0.000018903856,0.73792064,0.000543938,0.26040432,0.0007945613,0.000017563825],"about_ca_topic_score_codex":0.0028697965,"about_ca_topic_score_gemma":0.0020809628,"teacher_disagreement_score":0.009642614,"about_ca_system_score_codex":0.003545469,"about_ca_system_score_gemma":0.0024008935,"threshold_uncertainty_score":0.050995648},"labels":[],"label_agreement":null},{"id":"W4416429290","doi":"10.1109/tnnls.2025.3626050","title":"Universal Stabilization for Maximum Entropy Optimization in Reinforcement Learning","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Neural Networks and Learning Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"China Scholarship Council; National Natural Science Foundation of China","keywords":"Entropy (arrow of time); Kullback–Leibler divergence; Reinforcement learning; Principle of maximum entropy; Upper and lower bounds; Monotonic function; Maximum entropy spectral estimation; Divergence (linguistics)","score_opus":0.01099109538066441,"score_gpt":0.23014466149186183,"score_spread":0.2191535661111974,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416429290","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009750555,0.00022196243,0.98743546,0.00019873523,0.000026895632,0.000035898804,0.00001679052,0.0003156197,0.001998045],"genre_scores_gemma":[0.8543645,0.00031125057,0.14138764,0.00032827054,0.000074780684,0.00028557435,0.00009270924,0.0002674569,0.0028877964],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990841,0.00035173554,0.00005425868,0.000217425,0.00019349044,0.0000988854],"domain_scores_gemma":[0.9969607,0.0021540106,0.00032506467,0.00017585664,0.0002607667,0.00012355772],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024255684,0.0013019212,0.0012329714,0.0006599583,0.0006349146,0.0012372654,0.0010097486,0.0012408203,0.0020945948],"category_scores_gemma":[0.009910637,0.00060904515,0.00063944183,0.00042897346,0.0022354778,0.0014628106,0.0020806398,0.0020808806,0.00048199855],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006957455,0.000039447703,0.0005406508,0.000083547595,0.000035004232,0.00005526521,0.00008947573,0.94530743,0.0019696748,0.029643595,0.0007480304,0.021418327],"study_design_scores_gemma":[0.0000068949466,0.000020956977,0.000044947665,0.000009478508,0.0000035767475,0.0000073796878,0.0000042746296,0.9880547,0.00044285416,0.011180173,0.00021949293,0.000005227216],"about_ca_topic_score_codex":0.0024394835,"about_ca_topic_score_gemma":0.0019866535,"teacher_disagreement_score":0.0024394835,"about_ca_system_score_codex":0.001452046,"about_ca_system_score_gemma":0.0015874403,"threshold_uncertainty_score":0.012827754},"labels":[],"label_agreement":null},{"id":"W4416447818","doi":"10.48550/arxiv.2505.16217","title":"Reward-Aware Proto-Representations in Reinforcement Learning","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Alberta Innovates; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Representation (politics); Reinforcement learning; Space (punctuation); Function (biology); Key (lock); Successor cardinal; Encoding (memory)","score_opus":0.05062451241968759,"score_gpt":0.31850417763202704,"score_spread":0.26787966521233947,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416447818","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043216944,0.0003986736,0.9523542,0.00052543875,0.000039822422,0.000030601706,0.00011117805,0.0004911524,0.0028320488],"genre_scores_gemma":[0.8401144,0.0003136562,0.15598287,0.00019115597,0.0000416517,0.00014451698,0.00019929103,0.00010415602,0.0029082194],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99932146,0.00035168394,0.000027848771,0.00013433327,0.00010167256,0.00006302989],"domain_scores_gemma":[0.9975922,0.0015508145,0.00020488242,0.00034777666,0.00017776879,0.00012656201],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00163505,0.0004803226,0.0007965633,0.00046368368,0.00031075417,0.00096046174,0.0012866982,0.0010791087,0.0027012136],"category_scores_gemma":[0.0070302924,0.0003089245,0.00046492406,0.00060235296,0.0015526643,0.002580364,0.001106244,0.0019136409,0.00035638394],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018082657,0.00010253539,0.0009850343,0.00011929747,0.000032740885,0.00006802851,0.00017565317,0.6322062,0.001397998,0.28741315,0.001814591,0.075503916],"study_design_scores_gemma":[0.000013973552,0.000030883704,0.00006138368,0.000010189241,0.0000033902763,0.000012856048,0.000008926755,0.89873594,0.00018138796,0.1004834,0.00045140553,0.00000615984],"about_ca_topic_score_codex":0.0013126989,"about_ca_topic_score_gemma":0.0016666315,"teacher_disagreement_score":0.0027012136,"about_ca_system_score_codex":0.0011769773,"about_ca_system_score_gemma":0.0008444589,"threshold_uncertainty_score":0.009036481},"labels":[],"label_agreement":null},{"id":"W4416464763","doi":"10.1007/978-981-95-4367-0_40","title":"An Ensemble Method with Plans-Managed Policy for Proximal Policy Optimization","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Reinforcement learning; Ensemble learning; Policy learning; Optimization algorithm; Optimization problem; Ensemble forecasting","score_opus":0.015078387336756854,"score_gpt":0.29315041143452736,"score_spread":0.2780720240977705,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416464763","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0028811458,0.00016789905,0.9954117,0.000047881356,0.00007442344,0.00002408432,0.000020635369,0.00020419416,0.0011681055],"genre_scores_gemma":[0.23991583,0.00043702207,0.74966097,0.00023735974,0.00027383873,0.00052776746,0.0003095061,0.00041233256,0.008225389],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994281,0.0001958932,0.000032272037,0.000110280555,0.00016163386,0.00007175073],"domain_scores_gemma":[0.9987531,0.000746737,0.000056843433,0.00012387469,0.00022858758,0.000090978596],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014688855,0.000995671,0.0020210238,0.0007355511,0.00077613874,0.000877935,0.0022423195,0.002430715,0.006176454],"category_scores_gemma":[0.0031036735,0.0009666299,0.0011467471,0.001044899,0.0007961744,0.0013133122,0.0024752377,0.0024614118,0.0011514473],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008097655,0.00007162326,0.00022076626,0.00006891194,0.00007553115,0.000045684737,0.000039952454,0.87471837,0.0014046663,0.014417759,0.0022255688,0.10663022],"study_design_scores_gemma":[0.0000030027793,0.000010764581,0.0000131631805,0.000003131453,0.0000032427097,0.000004078759,0.0000013084436,0.9983646,0.00009353363,0.0013000276,0.00020095769,0.000002224768],"about_ca_topic_score_codex":0.0059562447,"about_ca_topic_score_gemma":0.0056363554,"teacher_disagreement_score":0.006176454,"about_ca_system_score_codex":0.0007215803,"about_ca_system_score_gemma":0.001510466,"threshold_uncertainty_score":0.020662248},"labels":[],"label_agreement":null},{"id":"W4416524723","doi":"10.48550/arxiv.2507.00275","title":"Deep Double Q-learning","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Alberta Innovates; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Deep learning; Double loop; Double-precision floating-point format; Carry (investment); Double layered; Double layer (biology); Key (lock)","score_opus":0.043713119429636856,"score_gpt":0.2866334611570382,"score_spread":0.24292034172740135,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416524723","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010269975,0.00023191955,0.9863344,0.000227436,0.00006627835,0.000065872104,0.00005238331,0.0010196571,0.0017320538],"genre_scores_gemma":[0.67023945,0.00020892841,0.32251874,0.0006731273,0.00008798107,0.00036868505,0.00034040675,0.00030820144,0.005254561],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99805933,0.00060546986,0.0001203244,0.0005821014,0.00037744353,0.0002553908],"domain_scores_gemma":[0.99456114,0.0029678545,0.00040367586,0.0007495268,0.0010408883,0.0002769092],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035962036,0.0012887997,0.0016024191,0.00056545244,0.00058710296,0.00134142,0.0027209646,0.00146289,0.0049066953],"category_scores_gemma":[0.0115811555,0.0006681793,0.0005962808,0.00054415077,0.001715895,0.0018985042,0.0025505642,0.0024716859,0.0010665135],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036713897,0.00021098323,0.0025074515,0.0002239704,0.00011472119,0.0000969257,0.00013937002,0.6802584,0.0036952249,0.03546547,0.005155716,0.27176458],"study_design_scores_gemma":[0.000023081027,0.000051766514,0.00008604182,0.0000097612365,0.0000064342466,0.000013710066,0.000005009348,0.98830146,0.00083429605,0.010034418,0.0006281091,0.0000059238837],"about_ca_topic_score_codex":0.004559655,"about_ca_topic_score_gemma":0.0046874518,"teacher_disagreement_score":0.0049066953,"about_ca_system_score_codex":0.0014148247,"about_ca_system_score_gemma":0.0026974075,"threshold_uncertainty_score":0.01901877},"labels":[],"label_agreement":null},{"id":"W4416748794","doi":"10.1109/iros60139.2025.11246593","title":"RA-DP: Rapid Adaptive Diffusion Policy for Training-Free High-frequency Robotics Replanning","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Huawei Technologies (Canada)","funders":"","keywords":"Robotics; Scalability; Reinforcement learning; Task (project management); Robot; Adaptation (eye); Range (aeronautics); Adaptive sampling; Sampling (signal processing)","score_opus":0.037698156261299624,"score_gpt":0.2881238452098909,"score_spread":0.2504256889485913,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416748794","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0068502673,0.00021154506,0.9912547,0.00010707116,0.00003744458,0.000034070537,0.000015459746,0.00063864904,0.0008508364],"genre_scores_gemma":[0.72785395,0.00035165995,0.2670778,0.00028536443,0.0000618656,0.00026693396,0.0001137148,0.00023808212,0.0037505706],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994548,0.00012530448,0.000031445936,0.00012348894,0.000193077,0.000071873175],"domain_scores_gemma":[0.99885726,0.0006082635,0.00013672118,0.00011782687,0.00018414417,0.000095883785],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012392894,0.000891148,0.0010277012,0.00042623226,0.0003890944,0.00054477504,0.0021380363,0.00116539,0.0015403515],"category_scores_gemma":[0.003783863,0.00047783426,0.000503124,0.00031900342,0.00088158436,0.0010011914,0.001357598,0.0018827843,0.0003787222],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009250577,0.00007349294,0.00042285334,0.00012138949,0.000036004214,0.000068498,0.00008847045,0.8902602,0.007874543,0.009217461,0.0016542285,0.09009026],"study_design_scores_gemma":[0.000012373008,0.000028976929,0.000037447327,0.0000047039466,0.0000039370398,0.000017262368,0.00000396496,0.99686384,0.00094962324,0.0015489953,0.0005237193,0.000005174756],"about_ca_topic_score_codex":0.0040574274,"about_ca_topic_score_gemma":0.0035219404,"teacher_disagreement_score":0.0040574274,"about_ca_system_score_codex":0.0006843256,"about_ca_system_score_gemma":0.0016081532,"threshold_uncertainty_score":0.0080676675},"labels":[],"label_agreement":null},{"id":"W4416749759","doi":"10.1109/iros60139.2025.11246340","title":"Generalizable Humanoid Manipulation with 3D Diffusion Policies","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Humanoid robot; Teleoperation; Robot; Robot control; Robotics; Telerobotics","score_opus":0.019004686340609908,"score_gpt":0.2618350488401887,"score_spread":0.2428303624995788,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416749759","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.35237232,0.00048434795,0.6394754,0.00040066047,0.000083988605,0.0001654868,0.00020854831,0.0022471042,0.0045621353],"genre_scores_gemma":[0.95915955,0.00005371091,0.03943521,0.00007940594,0.000006419689,0.00009734633,0.00015314494,0.00007634522,0.00093896384],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99966073,0.00010815403,0.000019041554,0.00007899478,0.00007025446,0.00006279481],"domain_scores_gemma":[0.9983955,0.0009968232,0.00016002433,0.00020047215,0.00014204546,0.000105110186],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012181288,0.0007418339,0.000723639,0.00035282725,0.00033637267,0.0004922959,0.0006279348,0.0007776657,0.0018892912],"category_scores_gemma":[0.004116766,0.00035034906,0.00046437213,0.00019141032,0.0008843224,0.0006725422,0.0010266966,0.0009988438,0.00026043586],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013588589,0.00006958412,0.0012767081,0.00004951553,0.000026409189,0.0000535119,0.000063718435,0.9742701,0.0028133558,0.0013339119,0.0005372059,0.019370139],"study_design_scores_gemma":[0.000015423218,0.000046080982,0.0002739931,0.000004953204,0.000003364581,0.000010633828,0.0000115811845,0.997324,0.0009048897,0.001135654,0.00026420417,0.000005274132],"about_ca_topic_score_codex":0.009069493,"about_ca_topic_score_gemma":0.0062360703,"teacher_disagreement_score":0.009069493,"about_ca_system_score_codex":0.00088156044,"about_ca_system_score_gemma":0.0009075548,"threshold_uncertainty_score":0.018033445},"labels":[],"label_agreement":null},{"id":"W4416756242","doi":"10.1109/lra.2025.3632119","title":"X-Nav: Learning End-to-End Cross-Embodiment Navigation for Mobile Robots","year":2025,"lang":"en","type":"article","venue":"IEEE Robotics and Automation Letters","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Mobile robot; Robot; Reinforcement learning; Robot learning; Robotics; Generalizability theory; Mobile robot navigation","score_opus":0.010608148763294734,"score_gpt":0.2849909912877431,"score_spread":0.27438284252444833,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416756242","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017215956,0.0001530059,0.97719574,0.00006193581,0.00006112178,0.00005348251,0.00004529134,0.0039360686,0.0012773178],"genre_scores_gemma":[0.6267643,0.00017683746,0.3667053,0.00026757867,0.000040684718,0.00022174983,0.00034889788,0.00041506684,0.005059552],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99975115,0.000045187535,0.000012759935,0.000094967,0.00005825648,0.000037684877],"domain_scores_gemma":[0.9996171,0.00015067676,0.000043889057,0.000078337915,0.0000732068,0.000036789843],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007160641,0.0010916864,0.0007123849,0.00020406162,0.0002770432,0.0004409027,0.0015055757,0.00080247846,0.0022262947],"category_scores_gemma":[0.0016722869,0.000475894,0.00042023798,0.00015475953,0.00064315455,0.0009116859,0.0015281746,0.0014626691,0.000600967],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024228389,0.0001348045,0.0013727973,0.00012042232,0.00007782134,0.00013907012,0.0001558504,0.69688,0.013534954,0.008099837,0.0035556094,0.27568647],"study_design_scores_gemma":[0.00001331006,0.00009340505,0.00010350579,0.000006885204,0.0000056861713,0.000023378425,0.000009425114,0.99318296,0.0024798084,0.0032443956,0.0008308441,0.000006349871],"about_ca_topic_score_codex":0.0036453288,"about_ca_topic_score_gemma":0.0052966243,"teacher_disagreement_score":0.0036453288,"about_ca_system_score_codex":0.00038012382,"about_ca_system_score_gemma":0.00096580415,"threshold_uncertainty_score":0.0074477196},"labels":[],"label_agreement":null},{"id":"W4416828434","doi":"10.1016/j.cogsys.2025.101422","title":"Dual or unified: optimizing drive-based reinforcement learning for cognitive autonomous robots","year":2025,"lang":"en","type":"article","venue":"Cognitive Systems Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Ministério da Ciência, Tecnologia e Inovação; Ministério da Ciência, Tecnologia, Inovações e Comunicações; Fundação de Amparo à Pesquisa do Estado de São Paulo; Conselho Nacional de Desenvolvimento Científico e Tecnológico; Brazilian Institute of Neuroscience and Neurotechnology","keywords":"Reinforcement learning; Modular design; Dual (grammatical number); Cognition; Curiosity; Selection (genetic algorithm); Robot; Autonomous agent; Action selection","score_opus":0.10680813833711963,"score_gpt":0.39253079557312515,"score_spread":0.28572265723600554,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416828434","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049399488,0.0003479618,0.94384676,0.0002718427,0.00009982135,0.00007382079,0.000044955606,0.00057729456,0.005337893],"genre_scores_gemma":[0.92093045,0.00010129035,0.07568845,0.00012698976,0.000039293136,0.00012563591,0.00004619018,0.00007309167,0.0028685005],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99958795,0.00012620293,0.000018905153,0.00008332901,0.00009856253,0.000085065934],"domain_scores_gemma":[0.99898654,0.0004946681,0.00010109703,0.00011549751,0.00019655295,0.00010564488],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013914037,0.000893518,0.0012559909,0.00040274812,0.00042159684,0.0009508038,0.001783544,0.0013122595,0.0027569404],"category_scores_gemma":[0.003734657,0.0005540649,0.00048703357,0.0003302522,0.0011619709,0.0012372661,0.0019238959,0.0015036617,0.00034579937],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003103782,0.00015868951,0.0006394901,0.000116437914,0.000072709845,0.00007774826,0.00009802503,0.91036445,0.0026544956,0.018704066,0.0016175582,0.065186],"study_design_scores_gemma":[0.000016851172,0.000050370145,0.000044974622,0.0000039850947,0.0000062612958,0.0000061102483,0.0000040026725,0.9943224,0.00026024086,0.005116531,0.00016461877,0.0000036062536],"about_ca_topic_score_codex":0.0038203818,"about_ca_topic_score_gemma":0.0043274257,"teacher_disagreement_score":0.0038203818,"about_ca_system_score_codex":0.00077835814,"about_ca_system_score_gemma":0.0012734636,"threshold_uncertainty_score":0.009222925},"labels":[],"label_agreement":null},{"id":"W4417094986","doi":"10.48550/arxiv.2502.16772","title":"Model-Based Exploration in Monitored Markov Decision Processes","year":2025,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates; University of Alberta; Alberta Machine Intelligence Institute; Alliance de recherche numérique du Canada; Mitacs; Canadian Institute for Advanced Research","keywords":"Markov decision process; Leverage (statistics); Exploit; Partially observable Markov decision process; Convergence (economics); Reinforcement learning; Process (computing); Markov process","score_opus":0.09140309050084275,"score_gpt":0.22014486556733345,"score_spread":0.1287417750664907,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417094986","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038052905,0.00038236874,0.9587674,0.0003586711,0.000022177446,0.000041404524,0.00007482548,0.0003590664,0.0019412353],"genre_scores_gemma":[0.9139949,0.00029174646,0.08374949,0.00010824929,0.000025005533,0.00017115548,0.00011541532,0.00005771869,0.0014862426],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9989002,0.0005447844,0.000044745455,0.0002125995,0.00016606352,0.00013158924],"domain_scores_gemma":[0.9923654,0.006260115,0.00059913576,0.0003063532,0.0002536966,0.00021511556],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002472923,0.00085349614,0.0012531643,0.0005243384,0.00040667446,0.0011327454,0.00123706,0.0010471626,0.0016915632],"category_scores_gemma":[0.01112184,0.00071781187,0.00066625356,0.0005403913,0.0015179752,0.0017293971,0.0014588637,0.0019380739,0.0001909484],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000053523043,0.000020932357,0.00052905007,0.00004188598,0.000018850185,0.000033246815,0.00004544465,0.9680734,0.00019745737,0.023787579,0.00020995126,0.0069886018],"study_design_scores_gemma":[0.000008917101,0.000012794102,0.000041310228,0.0000054489497,0.0000026628338,0.0000057351954,0.0000036077959,0.98170346,0.00009652969,0.018001372,0.00011557157,0.000002599996],"about_ca_topic_score_codex":0.004830432,"about_ca_topic_score_gemma":0.003809011,"teacher_disagreement_score":0.004830432,"about_ca_system_score_codex":0.0015419238,"about_ca_system_score_gemma":0.0016211956,"threshold_uncertainty_score":0.013078213},"labels":[],"label_agreement":null},{"id":"W4417239585","doi":"10.1038/s41467-025-66009-y","title":"Discovery of the reward function for embodied reinforcement learning agents","year":2025,"lang":"en","type":"article","venue":"Nature Communications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"National Key Research and Development Program of China; State Key Laboratory of Industrial Control Technology; State Key Laboratory of Mechanical Transmissions; Huazhong University of Science and Technology; National Natural Science Foundation of China","keywords":"Embodied cognition; Reinforcement learning; Regret; Maximization; Adaptability; Cognition; Function (biology); Cognitive robotics","score_opus":0.025178428916446973,"score_gpt":0.30649491024565473,"score_spread":0.2813164813292078,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417239585","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026709886,0.00014675714,0.96902746,0.0003296825,0.00002457859,0.000029196568,0.000015296046,0.00012874686,0.0035884578],"genre_scores_gemma":[0.7922278,0.00027338593,0.2036972,0.00013319243,0.000023822584,0.0001333251,0.000041292424,0.000085214866,0.0033847163],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996804,0.00009826229,0.000017807388,0.000060647606,0.000090641726,0.00005220851],"domain_scores_gemma":[0.99908686,0.0005141536,0.0001325352,0.000060699982,0.00012246834,0.000083335064],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010304748,0.00054989784,0.0007545646,0.00037514095,0.00039415495,0.0009040797,0.00075520226,0.0011613186,0.0013505766],"category_scores_gemma":[0.0043983064,0.0004415218,0.00047694985,0.00020374046,0.0012525283,0.000984089,0.0014273531,0.0013267827,0.0003099798],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000048051974,0.000039949937,0.00068608194,0.00007324508,0.000024150831,0.00010924784,0.000100004414,0.85861,0.0055633723,0.114201285,0.0006415738,0.019903043],"study_design_scores_gemma":[0.0000070226456,0.00001540901,0.00003937505,0.000006811101,0.0000031270629,0.000012396469,0.0000055349733,0.9823569,0.00060242735,0.016519409,0.00042704341,0.0000045828765],"about_ca_topic_score_codex":0.0011051628,"about_ca_topic_score_gemma":0.0009719761,"teacher_disagreement_score":0.0013505766,"about_ca_system_score_codex":0.0008194904,"about_ca_system_score_gemma":0.0010655696,"threshold_uncertainty_score":0.0059458613},"labels":[],"label_agreement":null},{"id":"W4417251985","doi":"10.1109/lra.2025.3643304","title":"An Intention-Guided Reinforcement Learning Approach With Dirichlet Energy Constraint for Heterogeneous Multi-Robot Cooperation","year":2025,"lang":"","type":"article","venue":"IEEE Robotics and Automation Letters","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"National Natural Science Foundation of China","keywords":"Reinforcement learning; Observability; Constraint (computer-aided design); Adaptability; Process (computing); Robot; Variety (cybernetics); Convergence (economics); Software deployment","score_opus":0.022887647404219565,"score_gpt":0.26651502927352133,"score_spread":0.24362738186930177,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417251985","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014517482,0.00017995178,0.98180765,0.00029736845,0.000047623394,0.000036773283,0.000027635268,0.0001804529,0.0029051208],"genre_scores_gemma":[0.88477254,0.00016922425,0.10851655,0.00027760913,0.00007613239,0.00025753808,0.000107633816,0.00008289826,0.0057398025],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993292,0.00024355791,0.000032125172,0.00014170425,0.00013598432,0.000117483],"domain_scores_gemma":[0.9985434,0.00087666145,0.00015028269,0.000087134365,0.00019792153,0.0001445005],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014140899,0.00096186507,0.0016304269,0.0005796736,0.0005176794,0.0008352876,0.0024226152,0.0014196044,0.0031260876],"category_scores_gemma":[0.002976852,0.0005680965,0.00077770086,0.0004695276,0.0012157909,0.0012300665,0.0017537002,0.0016819014,0.0004384266],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008086779,0.00007316351,0.00044664307,0.000049561455,0.000050120998,0.00014444011,0.00008169035,0.9539784,0.0010449376,0.01892776,0.0009937155,0.024128694],"study_design_scores_gemma":[0.000010386273,0.0000105736935,0.0000204173,0.000001929072,0.0000032533246,0.000004739377,0.0000032354408,0.9963452,0.000074838164,0.0034130611,0.00010891994,0.0000034155012],"about_ca_topic_score_codex":0.0061300746,"about_ca_topic_score_gemma":0.0044936333,"teacher_disagreement_score":0.0061300746,"about_ca_system_score_codex":0.0010050156,"about_ca_system_score_gemma":0.0013464919,"threshold_uncertainty_score":0.012188792},"labels":[],"label_agreement":null},{"id":"W4417293150","doi":"10.63282/3117-5481/aijcst-v3i5p102","title":"How Citizen Developers Changed the Game","year":2021,"lang":"","type":"article","venue":"American International Journal of Computer Science and Technology","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Microsemi (Canada)","funders":"","keywords":"Task (project management); Benchmarking; Deliverable; Modular design; Reinforcement learning; Action (physics); Autonomy; Software deployment; Intelligent agent","score_opus":0.012055228329681152,"score_gpt":0.2542565824229696,"score_spread":0.24220135409328847,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417293150","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1405237,0.0018857582,0.02421249,0.1790137,0.005998248,0.00030421698,0.00039266405,0.001701731,0.64596754],"genre_scores_gemma":[0.5921557,0.0012990569,0.01352622,0.02257266,0.0002580093,0.00019532088,0.00046278306,0.00207387,0.36745638],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.98919386,0.0059861233,0.00028366197,0.0013176048,0.001708858,0.0015098609],"domain_scores_gemma":[0.99138004,0.0024062612,0.00045207734,0.0011061471,0.002273758,0.0023816521],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007952656,0.0006048696,0.00034006263,0.0012329239,0.00964644,0.0122534875,0.0024371627,0.0044436604,0.024644515],"category_scores_gemma":[0.029705,0.00063323224,0.0006135728,0.0013419272,0.008866858,0.012347242,0.0077152043,0.006585337,0.008535767],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017143598,0.0003060975,0.012675876,0.0002069169,0.00003380723,0.0024465544,0.103517555,0.0007001632,0.0012959786,0.3098516,0.40732867,0.16146539],"study_design_scores_gemma":[0.000019761694,0.000037888516,0.0010065121,0.00011921361,0.0000090228505,0.00027024906,0.031402104,0.00060363003,0.0004044634,0.013987865,0.95210785,0.000031490315],"about_ca_topic_score_codex":0.026529852,"about_ca_topic_score_gemma":0.04165628,"teacher_disagreement_score":0.026529852,"about_ca_system_score_codex":0.0071963426,"about_ca_system_score_gemma":0.0074037113,"threshold_uncertainty_score":0.08244413},"labels":[],"label_agreement":null},{"id":"W4417338215","doi":"10.1109/icmlc66258.2025.11280109","title":"Learning to Communicate in Multi-Agent Reinforcement Learning for Autonomous Cyber Defence","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Defence Research and Development Canada; Royal Military College of Canada","funders":"","keywords":"Reinforcement learning; Battle; Autonomous agent; Error-driven learning; Intelligent agent; Game theory; Reinforcement; Limit (mathematics)","score_opus":0.04737860407282586,"score_gpt":0.31586286264509655,"score_spread":0.2684842585722707,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417338215","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.051480737,0.00036044154,0.94470924,0.00041641734,0.000046948004,0.00006105494,0.000017967503,0.00019616238,0.0027109704],"genre_scores_gemma":[0.9547726,0.00013480817,0.04279077,0.00010254492,0.000026762511,0.00014486798,0.000020363552,0.000024705618,0.0019825585],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99934334,0.00033778275,0.000031411437,0.00010537566,0.00010735997,0.00007471861],"domain_scores_gemma":[0.9982533,0.0011793999,0.00019898693,0.00007641678,0.00018620712,0.00010563521],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015634194,0.0007606518,0.0008933873,0.00027806172,0.00034872044,0.0006245191,0.0009225756,0.00090618577,0.0012388664],"category_scores_gemma":[0.004040452,0.00031548954,0.00035367388,0.00023555319,0.0012270176,0.00076484575,0.000985095,0.0012103987,0.00017323365],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004281073,0.000052513667,0.00035318095,0.00003503289,0.000020736514,0.00005127645,0.00005772362,0.9775263,0.000711584,0.010365752,0.00022976873,0.010553306],"study_design_scores_gemma":[0.000011379022,0.000023842958,0.00003369625,0.000002290414,0.0000025819418,0.0000040870314,0.0000034396012,0.9957295,0.00012555812,0.0039611533,0.00010019715,0.0000023038344],"about_ca_topic_score_codex":0.0035252138,"about_ca_topic_score_gemma":0.0025420527,"teacher_disagreement_score":0.0035252138,"about_ca_system_score_codex":0.0009256424,"about_ca_system_score_gemma":0.0009572351,"threshold_uncertainty_score":0.008268237},"labels":[],"label_agreement":null},{"id":"W4417531150","doi":"10.48550/arxiv.2512.16848","title":"Meta-RL Induces Exploration in Language Agents","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Canadian Institute for Advanced Research; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; National Science Foundation","keywords":"Reinforcement learning; Task (project management); Adaptation (eye); Generalization; Key (lock); Constructed language; Language model","score_opus":0.19891014669402876,"score_gpt":0.33922926549410454,"score_spread":0.14031911880007578,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417531150","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14364545,0.00039837637,0.84735477,0.00055981777,0.00006819837,0.000085169544,0.00008115968,0.0020987224,0.005708339],"genre_scores_gemma":[0.92415076,0.00009442004,0.07356457,0.0001770342,0.000018793995,0.00012793587,0.000067884044,0.00014786515,0.0016507115],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99928373,0.0003307133,0.000031006322,0.0001564442,0.000099043435,0.00009918891],"domain_scores_gemma":[0.9980762,0.0010014111,0.00023053319,0.00040444636,0.00015383265,0.00013348271],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001552595,0.00089357124,0.000809032,0.0003530298,0.0003739505,0.00087591837,0.001401515,0.0009106967,0.0016293251],"category_scores_gemma":[0.0056136874,0.0005678997,0.00064592925,0.00021342655,0.0016231549,0.001527274,0.0023180721,0.0014937517,0.00036893628],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015930846,0.00010110571,0.0020237474,0.00011936763,0.00008468558,0.00013557816,0.00020492905,0.9212615,0.011189141,0.019710496,0.0010216653,0.04398864],"study_design_scores_gemma":[0.000019428997,0.000058166133,0.00013231639,0.000008866548,0.000008761152,0.000019792295,0.000017206521,0.9872602,0.0017427913,0.010266419,0.00045855713,0.000007433012],"about_ca_topic_score_codex":0.0011460661,"about_ca_topic_score_gemma":0.001970226,"teacher_disagreement_score":0.0016293251,"about_ca_system_score_codex":0.0006584385,"about_ca_system_score_gemma":0.0009787434,"threshold_uncertainty_score":0.008211017},"labels":[],"label_agreement":null},{"id":"W4417538955","doi":"10.48550/arxiv.2506.16608","title":"Distributions as Actions: A Unified Framework for Diverse Action Spaces","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Alberta Machine Intelligence Institute; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Action (physics); Parameterized complexity; Simple (philosophy); Boundary (topology); Space (punctuation); Distribution (mathematics); Variance (accounting)","score_opus":0.12962069525450415,"score_gpt":0.36849466999100305,"score_spread":0.2388739747364989,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417538955","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0012515719,0.00012088671,0.99682546,0.00017420392,0.000026636248,0.00002011603,0.00003378846,0.0001453464,0.0014019675],"genre_scores_gemma":[0.3884069,0.0008684317,0.6014967,0.00038750522,0.00022052709,0.000510605,0.000250309,0.0004098384,0.0074491063],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99822634,0.0007039851,0.00009056212,0.00038926795,0.00046396532,0.00012588374],"domain_scores_gemma":[0.9979937,0.0009656249,0.0002443512,0.00032425963,0.0002603726,0.0002116487],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023943265,0.0012128304,0.0011261764,0.0008752246,0.0005041024,0.0024422817,0.0027852193,0.0014772121,0.004262675],"category_scores_gemma":[0.00711217,0.00071989855,0.0009680174,0.00091096526,0.0027252792,0.003722676,0.0029564518,0.0036640149,0.00094048894],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000059380567,0.000050636078,0.0004697477,0.00008724417,0.00004381911,0.00009117128,0.00015700131,0.5063908,0.0014843527,0.4435605,0.0019102796,0.045694996],"study_design_scores_gemma":[0.000017131777,0.00002849134,0.000059574984,0.00001780516,0.0000074885825,0.000025934438,0.000013235105,0.8667665,0.00034773737,0.12982075,0.0028845533,0.0000107849155],"about_ca_topic_score_codex":0.0032324644,"about_ca_topic_score_gemma":0.0033157065,"teacher_disagreement_score":0.004262675,"about_ca_system_score_codex":0.0017277349,"about_ca_system_score_gemma":0.0022811692,"threshold_uncertainty_score":0.014260113},"labels":[],"label_agreement":null},{"id":"W627618803","doi":"10.71781/10765","title":"Modèle informatique du coapprentissage des ganglions de la base et du cortex : l'apprentissage par renforcement et le développement de représentations","year":2009,"lang":"fr","type":"dissertation","venue":"Open MIND","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Canadian Institutes of Health Research","keywords":"Humanities; Political science; Philosophy","score_opus":0.052661862367999884,"score_gpt":0.3434632231230554,"score_spread":0.29080136075505547,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W627618803","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06865353,0.00640527,0.8881553,0.0047771484,0.00056991924,0.00008634572,0.00075790106,0.00085150544,0.029743094],"genre_scores_gemma":[0.84753937,0.007376885,0.1030612,0.00053643546,0.00032272283,0.0002719092,0.0006531714,0.00016952182,0.04006879],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996113,0.000087685694,0.000021434658,0.00012131731,0.000097296004,0.000060971583],"domain_scores_gemma":[0.9988697,0.00059300073,0.00013347753,0.000112915695,0.00021185914,0.00007907151],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007355445,0.0008450715,0.0009103488,0.00064684753,0.0005481866,0.0027348748,0.0015956728,0.0020753597,0.00679937],"category_scores_gemma":[0.0032504622,0.0005760852,0.001179215,0.00075600116,0.0014621685,0.0029327034,0.0011396968,0.0025068247,0.0011912725],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035626566,0.00009293666,0.0044734487,0.00048671046,0.00022479886,0.0005940761,0.00059664156,0.60656995,0.011778458,0.2696888,0.0058318013,0.09930602],"study_design_scores_gemma":[0.000029541465,0.00012241071,0.0018977722,0.00008894272,0.00008882693,0.00028302296,0.00011491584,0.8923696,0.0028487644,0.091488436,0.010613247,0.00005442998],"about_ca_topic_score_codex":0.011427401,"about_ca_topic_score_gemma":0.007551028,"teacher_disagreement_score":0.011427401,"about_ca_system_score_codex":0.0015447844,"about_ca_system_score_gemma":0.0020043277,"threshold_uncertainty_score":0.022746146},"labels":[],"label_agreement":null},{"id":"W64134055","doi":"","title":"Bayesian Learning of Recursively Factored Environments","year":2013,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Factoring; Computer science; Reinforcement learning; Factorization; Inference; Class (philosophy); Artificial intelligence; Bayesian probability; Task (project management); Machine learning; Scaling; Bayesian inference; Theoretical computer science; Algorithm; Mathematics","score_opus":0.00994807483159242,"score_gpt":0.20887989202429472,"score_spread":0.1989318171927023,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W64134055","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.057828348,0.0003252066,0.93749326,0.00027882977,0.000022869546,0.00006157395,0.00011957044,0.0005138094,0.003356546],"genre_scores_gemma":[0.8995642,0.00032206115,0.09670236,0.00009128817,0.000024337998,0.00012583358,0.00022571879,0.00010455606,0.0028395965],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986179,0.0007093646,0.00004276796,0.00027683272,0.00017842466,0.00017477313],"domain_scores_gemma":[0.9962863,0.0024401401,0.00041751406,0.00029621518,0.00032594608,0.00023394568],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023167368,0.0013313815,0.0015298089,0.0007378957,0.0005877836,0.0012314164,0.0016107223,0.0013664406,0.0031512205],"category_scores_gemma":[0.011136398,0.00093381637,0.00081204885,0.00050752965,0.0021997928,0.0032072864,0.0019115127,0.002167891,0.00054964965],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000113839946,0.000039821545,0.0009782142,0.000045843277,0.000033198674,0.00005844079,0.00011078083,0.93564993,0.00065278186,0.04059834,0.00073788204,0.02098091],"study_design_scores_gemma":[0.000015836165,0.000023045903,0.00014592211,0.000008790792,0.0000042799115,0.000009225682,0.000008622358,0.96359277,0.00012074452,0.03585472,0.00020798102,0.000008124898],"about_ca_topic_score_codex":0.012951774,"about_ca_topic_score_gemma":0.013363477,"teacher_disagreement_score":0.012951774,"about_ca_system_score_codex":0.0016665363,"about_ca_system_score_gemma":0.0014671758,"threshold_uncertainty_score":0.025752723},"labels":[],"label_agreement":null},{"id":"W6888783548","doi":"10.22067/econg.2023.80748.1066","title":"REE-Y-Ti-Th mineralization in the albite bearing metasomatite and metasomatized rhyolite hosted in Choghart magnetite-apatite deposit, Central Iran: Interplay of evaporitic brines, hydrothermal-magmatic fluids","year":2023,"lang":"en","type":"article","venue":"DOAJ (DOAJ: Directory of Open Access Journals)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Metasomatism; Mineralization (soil science); Hydrothermal circulation; Albite; Calcite; Rhyolite; Fluid inclusions; Isotope geochemistry; δ34S","score_opus":0.11849839901669237,"score_gpt":0.47022455640057076,"score_spread":0.35172615738387836,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6888783548","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99963295,0.000148743,0.00002772258,0.0000033936062,7.9270376e-7,0.0000024611597,0.000043497188,0.0000022058452,0.00013807186],"genre_scores_gemma":[0.9994336,0.000098222095,0.00011871666,0.0000037469085,0.0000015627721,0.0000037204238,0.0000878327,0.0000016788915,0.0002508553],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9999411,0.000004555918,0.0000055314194,0.000021191281,0.000013101681,0.000014528013],"domain_scores_gemma":[0.9999329,0.0000068157983,0.000023901106,0.0000029480802,0.000020506555,0.0000129174705],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00010563376,0.00024191178,0.00017763958,0.0008848293,0.00022367043,0.00039130283,0.00026067707,0.00027313252,0.00029420893],"category_scores_gemma":[0.00009638085,0.00027287612,0.00020056251,0.0003284834,0.0003127428,0.00020694018,0.00019482744,0.00013552298,0.00009912153],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005055167,0.000054949753,0.16044027,0.00013611911,0.000057348334,0.0006667389,0.0006485076,0.00020260828,0.8315965,0.00008259704,0.00003547457,0.0055735456],"study_design_scores_gemma":[0.00001605664,0.00023566319,0.9621985,0.0000055107284,0.00004073489,0.0007257795,0.000555329,0.000521932,0.035141453,0.00006128728,0.0004888301,0.000008928015],"about_ca_topic_score_codex":0.010392708,"about_ca_topic_score_gemma":0.012862846,"teacher_disagreement_score":0.010392708,"about_ca_system_score_codex":0.00032221124,"about_ca_system_score_gemma":0.0002917545,"threshold_uncertainty_score":0.020664394},"labels":[],"label_agreement":null},{"id":"W6892327132","doi":"10.5065/d6f47m5p","title":"SWL11 Bottle data. Version 1.0","year":2015,"lang":"en","type":"dataset","venue":"Earth Observing Laboratory","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Fisheries and Oceans Canada","funders":"","keywords":"Coast guard; Cruise; Hydrography; Water bottle; Longitude; Bottle; Latitude; Colored dissolved organic matter","score_opus":0.05144617681186247,"score_gpt":0.2781954881882576,"score_spread":0.22674931137639515,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6892327132","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00032056335,0.00006458539,0.000102619255,0.000046538586,0.00003123157,0.000018951941,0.9979875,0.0009212492,0.0005066514],"genre_scores_gemma":[0.00027410817,0.00002757725,0.00025784754,0.000025420233,0.000003953856,0.000048273687,0.99897206,0.000059062564,0.0003317674],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99863297,0.00018606774,0.0001818688,0.0004187135,0.00035657216,0.00022376773],"domain_scores_gemma":[0.99873644,0.00023493425,0.000104853585,0.00036894385,0.00040189008,0.00015287288],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011044109,0.0038164498,0.0017753915,0.0031363382,0.0010002523,0.0020783779,0.004199812,0.002430086,0.034653094],"category_scores_gemma":[0.0037727226,0.00090074446,0.0019133048,0.0054901964,0.0005629969,0.0018562601,0.0024171316,0.0020071208,0.08064358],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007460247,0.00004678478,0.0009959615,0.0006029336,0.00003884165,0.00003099554,0.000027631439,0.00053051935,0.0002717159,0.00025404163,0.9939236,0.0032023545],"study_design_scores_gemma":[0.00029431068,0.000055044886,0.005682333,0.00021786292,0.000044365457,0.00011587906,0.00013924541,0.0017570893,0.0010573976,0.0010062188,0.9895635,0.000066688975],"about_ca_topic_score_codex":0.036327355,"about_ca_topic_score_gemma":0.07062808,"teacher_disagreement_score":0.036327355,"about_ca_system_score_codex":0.0014799852,"about_ca_system_score_gemma":0.0027097783,"threshold_uncertainty_score":0.11592615},"labels":[],"label_agreement":null},{"id":"W6893087174","doi":"10.5281/zenodo.14162773","title":"Hydrolysate Untargeted Metabolomics Mass Spectrometry Data","year":2024,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Hydrolysate; Mass spectrometry; Metabolomics; Liquid chromatography–mass spectrometry; Analyte","score_opus":0.0442405478327345,"score_gpt":0.2672488440427971,"score_spread":0.22300829621006257,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6893087174","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0033550428,0.00038782862,0.00061252725,0.000107500186,0.000057521225,0.0000411226,0.99347985,0.0010387617,0.0009197894],"genre_scores_gemma":[0.0025536309,0.0001136299,0.0009969014,0.000055873476,0.000008206891,0.00007424982,0.9954591,0.00007772911,0.0006607214],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99904543,0.00010582657,0.00006536483,0.00034756324,0.00029292138,0.00014277673],"domain_scores_gemma":[0.9991899,0.00015997978,0.00008984517,0.00027461827,0.0001894719,0.000096173004],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010642299,0.003649961,0.0018015665,0.0030768788,0.0008614472,0.0014881868,0.002718722,0.003286027,0.017469924],"category_scores_gemma":[0.001986091,0.0006021604,0.0020893805,0.0034698728,0.0007907938,0.0006775125,0.0015932643,0.0017019915,0.027432995],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017339651,0.00045624224,0.004638765,0.0033020906,0.00051032624,0.0005336231,0.00006856513,0.004042601,0.018126626,0.0010286134,0.94605553,0.019503033],"study_design_scores_gemma":[0.0013390925,0.00027488804,0.024573969,0.0002928041,0.00046900893,0.0009897588,0.00013322172,0.005583302,0.01986546,0.003952047,0.94234955,0.00017680715],"about_ca_topic_score_codex":0.012515273,"about_ca_topic_score_gemma":0.02199627,"teacher_disagreement_score":0.017469924,"about_ca_system_score_codex":0.0010606191,"about_ca_system_score_gemma":0.0018954517,"threshold_uncertainty_score":0.058442652},"labels":[],"label_agreement":null},{"id":"W6894416521","doi":"10.5555/3635637.3662926","title":"MaDi:Learning to Mask Distractions for Generalization in Visual Deep Reinforcement Learning","year":2024,"lang":"en","type":"article","venue":"TU/e Research Portal","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Generalization; Reinforcement learning; Focus (optics); Artificial neural network; Contrast (vision); Control (management); Deep learning","score_opus":0.05700158475965024,"score_gpt":0.4050078222895636,"score_spread":0.3480062375299134,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6894416521","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06578226,0.00070259004,0.920484,0.00046295693,0.00013539684,0.00015263866,0.00014042316,0.008205905,0.0039338074],"genre_scores_gemma":[0.77701116,0.00015506652,0.21719041,0.00048598353,0.000050107974,0.00023596793,0.0003183734,0.0003234799,0.004229381],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996697,0.00007056948,0.000016835864,0.000098721204,0.00007951543,0.00006473917],"domain_scores_gemma":[0.9992514,0.0003207206,0.000103340026,0.00015383045,0.00008953284,0.00008098591],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013534043,0.0014219425,0.0008520733,0.00032556895,0.00032177565,0.00063281046,0.0022331923,0.0011197376,0.0021958882],"category_scores_gemma":[0.003458687,0.0004453235,0.0004656927,0.00019095605,0.0008573013,0.0010359806,0.0015610884,0.0022739582,0.00045083763],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030161286,0.0002961404,0.0022644082,0.00015274547,0.00011134673,0.00009601819,0.00014494665,0.6897437,0.013778207,0.008112213,0.006947666,0.27805114],"study_design_scores_gemma":[0.000025477642,0.00007633557,0.00012413407,0.000008201995,0.000007831199,0.000016126054,0.000005551955,0.99404967,0.002830026,0.0023143387,0.0005370655,0.000005149254],"about_ca_topic_score_codex":0.0041261604,"about_ca_topic_score_gemma":0.0051663853,"teacher_disagreement_score":0.0041261604,"about_ca_system_score_codex":0.0011241377,"about_ca_system_score_gemma":0.0012587957,"threshold_uncertainty_score":0.008204281},"labels":[],"label_agreement":null},{"id":"W6901636194","doi":"10.60692/16y90-1d376","title":"Editorial: Advances in Robots Trajectories Learning via Fast Neural Networks","year":2021,"lang":"en","type":"article","venue":"Greater South Information System","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Artificial neural network; Robot; Trajectory; Feature (linguistics); Key (lock)","score_opus":0.01361408417350617,"score_gpt":0.21204383618730055,"score_spread":0.19842975201379437,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6901636194","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00005734229,0.0042569023,0.00032217952,0.01957715,0.9733848,0.000023689274,0.00012284462,0.00010727192,0.002147758],"genre_scores_gemma":[0.0007954329,0.0036377497,0.00017496786,0.009522923,0.97338253,0.000025717201,0.0000620078,0.00006124806,0.012337431],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99758077,0.00032529232,0.00029503292,0.00045056661,0.0011322374,0.0002160272],"domain_scores_gemma":[0.9870903,0.0050877277,0.0007734128,0.0003451663,0.0048062084,0.0018970855],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004493482,0.0036759498,0.003164043,0.0029722897,0.0020496722,0.005586682,0.002960994,0.010573277,0.0364391],"category_scores_gemma":[0.015729202,0.00093505566,0.002340845,0.00107464,0.0017703824,0.003274699,0.0011535671,0.010517369,0.021265388],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008253508,0.000014705915,0.000023320865,0.00025050066,0.000022948276,0.00007738716,0.000005417299,0.00006260068,0.00009666313,0.00041835863,0.99228656,0.0066590216],"study_design_scores_gemma":[0.000116769545,0.000070894384,0.00029839537,0.00030158245,0.00007610058,0.00021773251,0.000019500641,0.00045263473,0.0003005941,0.0018053552,0.9963134,0.000027065209],"about_ca_topic_score_codex":0.0007239118,"about_ca_topic_score_gemma":0.0016800994,"teacher_disagreement_score":0.0364391,"about_ca_system_score_codex":0.0016414946,"about_ca_system_score_gemma":0.0013738712,"threshold_uncertainty_score":0.121900916},"labels":[],"label_agreement":null},{"id":"W6908170018","doi":"10.25573/serc.c.6100506.v5","title":"MarineGEO Bocas del Toro Observatory Data","year":2022,"lang":"en","type":"other","venue":"Figshare","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Observatory; Nova scotia; Context (archaeology)","score_opus":0.11380987495605834,"score_gpt":0.2849537580977909,"score_spread":0.17114388314173257,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6908170018","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00799074,0.00016974477,0.0012393184,0.000587276,0.00026334976,0.00013697773,0.9451281,0.00133453,0.043149903],"genre_scores_gemma":[0.058874372,0.0003147582,0.009398167,0.0002940105,0.00013700665,0.0005657716,0.89040387,0.0013649088,0.038647212],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.9996038,0.0000379001,0.000018369916,0.00010138346,0.00017422727,0.00006441572],"domain_scores_gemma":[0.9990482,0.000088617664,0.000101062295,0.00018931244,0.00042891077,0.00014391643],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00045966663,0.0006069364,0.00044790874,0.0014661393,0.0006170271,0.0007649062,0.0009831599,0.00055663614,0.061241623],"category_scores_gemma":[0.0019454003,0.00021444503,0.00025806078,0.002290077,0.00026186605,0.00051439146,0.0010144009,0.0007982803,0.026851071],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012681687,0.00004809456,0.013852588,0.00025717387,0.000025618892,0.00008941054,0.00011750777,0.0014721983,0.0006858042,0.0012580674,0.9593963,0.022670507],"study_design_scores_gemma":[0.00010333887,0.000018866793,0.040079705,0.00010212032,0.000019626987,0.000047346828,0.00023253686,0.0019666299,0.00057304447,0.0012821157,0.95554805,0.000026612355],"about_ca_topic_score_codex":0.10404904,"about_ca_topic_score_gemma":0.17184578,"teacher_disagreement_score":0.10404904,"about_ca_system_score_codex":0.00072403566,"about_ca_system_score_gemma":0.0013712196,"threshold_uncertainty_score":0.20688677},"labels":[],"label_agreement":null},{"id":"W6913045718","doi":"10.5555/3545946.3598862","title":"Automatic Noise Filtering with Dynamic Sparse Training in Deep Reinforcement Learning","year":2023,"lang":"en","type":"article","venue":"Open Repository and Bibliography (University of Luxembourg)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs; University of Alberta; Alberta Machine Intelligence Institute; Compute Canada; Nederlandse Organisatie voor Wetenschappelijk Onderzoek; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Noise (video); Focus (optics); Robot; Code (set theory); Training (meteorology); Transfer of learning; Deep learning","score_opus":0.018197541683775263,"score_gpt":0.22543555722018838,"score_spread":0.2072380155364131,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6913045718","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01603318,0.00024944457,0.9807176,0.00022158845,0.000034406723,0.000041442712,0.000056530305,0.0012640958,0.0013816272],"genre_scores_gemma":[0.7433433,0.00022923062,0.25270882,0.00028836247,0.000054056214,0.0002456456,0.00025111507,0.00021021826,0.0026692734],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995529,0.00013934307,0.000024137144,0.00009921308,0.000121384524,0.000062873245],"domain_scores_gemma":[0.9985588,0.00091028813,0.00013251277,0.00015350715,0.00017681021,0.00006807681],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013142757,0.0007869671,0.00080444035,0.00030926513,0.00029554995,0.0005910683,0.001285494,0.0008635185,0.0018098974],"category_scores_gemma":[0.0052716606,0.0004450114,0.00039932138,0.00033900436,0.0010760177,0.0009376835,0.0011201074,0.0017708727,0.00039070737],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008692899,0.00007557305,0.00076319964,0.00007897044,0.0000345391,0.00004458155,0.00005377459,0.90427125,0.0025347236,0.01028422,0.0014509534,0.08032132],"study_design_scores_gemma":[0.000010563973,0.000017122868,0.00003831072,0.0000051000698,0.0000029464852,0.0000052640157,0.0000022010563,0.9954045,0.0005640005,0.0036861289,0.0002614849,0.000002431967],"about_ca_topic_score_codex":0.0061433157,"about_ca_topic_score_gemma":0.0073554185,"teacher_disagreement_score":0.0061433157,"about_ca_system_score_codex":0.0009185567,"about_ca_system_score_gemma":0.0013602446,"threshold_uncertainty_score":0.012215078},"labels":[],"label_agreement":null},{"id":"W6918060949","doi":"10.58079/sgu","title":"Journée des doctorants de l’Association française d’études canadiennes","year":2015,"lang":"fr","type":"other","venue":"OpenEdition (OpenEdition)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Context (archaeology); Perspective (graphical); Subject (documents)","score_opus":0.03010377845884565,"score_gpt":0.2518991097946293,"score_spread":0.22179533133578366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6918060949","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03093939,0.03632486,0.005911918,0.55476093,0.060069233,0.00029769473,0.001806186,0.00057668344,0.3093131],"genre_scores_gemma":[0.09114516,0.009506457,0.0031038777,0.012611766,0.0041722013,0.00012311726,0.0006660832,0.00030167375,0.8783697],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99318886,0.00096408953,0.0001545032,0.0006635759,0.002750316,0.0022786441],"domain_scores_gemma":[0.9768063,0.001212768,0.00052240404,0.0005078845,0.005432135,0.015518576],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006823583,0.00053200027,0.0006189434,0.0016398139,0.010735851,0.0074259965,0.0012729245,0.004293825,0.07453532],"category_scores_gemma":[0.010164577,0.00042546907,0.000682393,0.0018322939,0.0033373383,0.0026414492,0.0049030622,0.0062862965,0.013950422],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001170771,0.00009678275,0.003938655,0.000121892175,0.00001487956,0.0003038264,0.0036375085,0.00021975438,0.00057362724,0.041354667,0.8681395,0.08148189],"study_design_scores_gemma":[0.000005913621,0.000011809282,0.0026070033,0.00006141519,0.0000025342715,0.00005852855,0.0010131829,0.000058360114,0.00014256558,0.0005935926,0.9954335,0.000011636146],"about_ca_topic_score_codex":0.45523658,"about_ca_topic_score_gemma":0.692896,"teacher_disagreement_score":0.97699,"about_ca_system_score_codex":0.02301,"about_ca_system_score_gemma":0.0736742,"threshold_uncertainty_score":0.9051736},"labels":[],"label_agreement":null},{"id":"W6930229253","doi":"10.5281/zenodo.11094777","title":"Signals and Systems OER Labs","year":2024,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Set (abstract data type); Key (lock); Data collection; Identification (biology)","score_opus":0.03538107557583014,"score_gpt":0.24977624318281305,"score_spread":0.2143951676069829,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6930229253","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0014158211,0.0018314687,0.09147957,0.0016889457,0.0013721548,0.00037346786,0.04473179,0.0775907,0.7795161],"genre_scores_gemma":[0.007877683,0.0020078071,0.02928826,0.00051178894,0.000750578,0.00024887404,0.04920955,0.015232865,0.89487255],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99796873,0.0001694341,0.00009263969,0.0002570943,0.0013715586,0.00014052463],"domain_scores_gemma":[0.9964097,0.0005151497,0.000118131386,0.0011243484,0.0012110746,0.00062152784],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0023236324,0.0023962578,0.0015818186,0.0032598192,0.0011491457,0.004633752,0.002770643,0.0014229016,0.6699377],"category_scores_gemma":[0.0046897708,0.00082020985,0.0008139779,0.0036933073,0.0005222865,0.0034963244,0.0030239625,0.0025844048,0.5763206],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008012296,0.00012421753,0.0001230288,0.00029417477,0.000011574069,0.00005781066,0.000041090574,0.0007298891,0.0022461903,0.01135045,0.79843956,0.18650183],"study_design_scores_gemma":[0.000043304262,0.00005150332,0.00022021795,0.000060868315,0.0000074992613,0.00008864941,0.000022339775,0.0021524557,0.0029187093,0.0064090253,0.98800373,0.00002159879],"about_ca_topic_score_codex":0.0016799674,"about_ca_topic_score_gemma":0.0031504491,"teacher_disagreement_score":0.6699377,"about_ca_system_score_codex":0.00096995337,"about_ca_system_score_gemma":0.0014788024,"threshold_uncertainty_score":0.47079384},"labels":[],"label_agreement":null},{"id":"W6949081508","doi":"10.5281/zenodo.12124482","title":"Iso 9110 pdf","year":2024,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Certification; Quality management system; Process (computing); Quality (philosophy); Quality of analytical results; Control (management); Risk management; Quality assurance","score_opus":0.027056946696764745,"score_gpt":0.2462997047227309,"score_spread":0.21924275802596616,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6949081508","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001111819,0.0016611132,0.041546293,0.0019689172,0.004198879,0.0018075488,0.04088355,0.0099404,0.8968816],"genre_scores_gemma":[0.0097432025,0.003286078,0.051345374,0.0026810945,0.00073521666,0.0017467682,0.14719865,0.007813466,0.7754503],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9866947,0.0011675096,0.0009984114,0.00068876566,0.0097605195,0.0006901346],"domain_scores_gemma":[0.985436,0.0005986825,0.00033247456,0.0012067935,0.01220385,0.00022227755],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.004741128,0.0025848784,0.00096054137,0.0072892117,0.0017280262,0.0061837784,0.003914671,0.003542283,0.24979025],"category_scores_gemma":[0.013189564,0.0011718425,0.0012647924,0.0063154506,0.0011180759,0.0077964882,0.0027507453,0.0031968758,0.31564468],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005389308,0.00007438558,0.00022453527,0.00048354716,0.000009394985,0.00006930933,0.00010893952,0.0006046332,0.0015304515,0.014115176,0.8737841,0.10894158],"study_design_scores_gemma":[0.000005541474,0.0000134775955,0.00023777079,0.00011008242,0.0000032619266,0.00004557247,0.00003985868,0.00012246874,0.0004709869,0.0009945626,0.99794465,0.000011680773],"about_ca_topic_score_codex":0.022362603,"about_ca_topic_score_gemma":0.014511808,"teacher_disagreement_score":0.75020975,"about_ca_system_score_codex":0.0031705976,"about_ca_system_score_gemma":0.007954067,"threshold_uncertainty_score":0.8356316},"labels":[],"label_agreement":null},{"id":"W6979290938","doi":"","title":"A Smooth Sea Never Made a Skilled $\\texttt{SAILOR}$: Robust Imitation via Learning to Search","year":2025,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Office of Naval Research; Canada Excellence Research Chairs, Government of Canada; National Science Foundation","keywords":"Mistake; Set (abstract data type); Imitation; Outcome (game theory); Code (set theory); Task (project management); Plan (archaeology); Process (computing); Test (biology); Component (thermodynamics); Expert system","score_opus":0.04761884207748829,"score_gpt":0.2007396210769313,"score_spread":0.153120778999443,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6979290938","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21969976,0.00215783,0.72574127,0.001165127,0.00027495436,0.00023708607,0.0006820266,0.034332618,0.015709367],"genre_scores_gemma":[0.8089317,0.0003419977,0.17785722,0.00036110735,0.00004592112,0.00023929251,0.001209125,0.0013989878,0.0096147135],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99928975,0.00017188219,0.000035634097,0.00025300222,0.00014336855,0.00010625763],"domain_scores_gemma":[0.99783933,0.0011207206,0.00016666288,0.00052085856,0.00021024347,0.00014220875],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011584433,0.0013445581,0.0011055608,0.0004756573,0.00045825346,0.0009033896,0.0030205029,0.001586061,0.004388047],"category_scores_gemma":[0.00647841,0.0006225399,0.0006944583,0.00034028495,0.0012623564,0.001901082,0.0018615705,0.002459558,0.0019944059],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00050159974,0.00035350566,0.002896142,0.0004888803,0.00014325386,0.00025769934,0.0002354451,0.61366844,0.018855209,0.0099199675,0.012473652,0.34020618],"study_design_scores_gemma":[0.000030471834,0.00010953077,0.00029119704,0.000018060198,0.0000118128855,0.000039951654,0.000021430107,0.9909561,0.0037525974,0.0034312052,0.0013245756,0.000012983794],"about_ca_topic_score_codex":0.010088115,"about_ca_topic_score_gemma":0.008815282,"teacher_disagreement_score":0.010088115,"about_ca_system_score_codex":0.00096733705,"about_ca_system_score_gemma":0.0014786236,"threshold_uncertainty_score":0.02005881},"labels":[],"label_agreement":null},{"id":"W6982999955","doi":"","title":"Legion of French Volunteers Against Bolshevism","year":2016,"lang":"en","type":"article","venue":"Digital Kenyon (Kenyon College)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Infantry; State (computer science); German; Government (linguistics); Head (geology); First world war","score_opus":0.009506763513425517,"score_gpt":0.20875069457741818,"score_spread":0.19924393106399266,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6982999955","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05803797,0.03076137,0.0008259768,0.15374053,0.019246135,0.00015909028,0.0015680952,0.0012512471,0.73440963],"genre_scores_gemma":[0.1387981,0.002054977,0.00030849784,0.009827534,0.0016731112,0.000057302066,0.0003339843,0.00015588471,0.8467907],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988103,0.00026229603,0.000024897703,0.00016451269,0.00036918285,0.00036878962],"domain_scores_gemma":[0.99857545,0.00023882448,0.00013942471,0.00008064652,0.0003674953,0.00059824594],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011754681,0.00075264025,0.00034842236,0.0010686569,0.007262042,0.0034618457,0.0004246633,0.0027244904,0.086681925],"category_scores_gemma":[0.0034206845,0.00030363558,0.0002798268,0.0005086266,0.0015387535,0.0010086404,0.0025331383,0.002170243,0.014548624],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018069624,0.00003212146,0.0021659958,0.000092055634,0.000007629894,0.00065337546,0.004228236,0.00006661951,0.0009443373,0.017123638,0.91720474,0.057300527],"study_design_scores_gemma":[0.0000040900777,0.000018034481,0.0016419285,0.00003300108,8.840623e-7,0.00009143464,0.00081160676,0.000018551198,0.00010875283,0.000098294695,0.9971674,0.000005943189],"about_ca_topic_score_codex":0.09670331,"about_ca_topic_score_gemma":0.1381896,"teacher_disagreement_score":0.09670331,"about_ca_system_score_codex":0.004617863,"about_ca_system_score_gemma":0.0030206365,"threshold_uncertainty_score":0.28997993},"labels":[],"label_agreement":null},{"id":"W6987579751","doi":"","title":"Temporal credit assignment via traces in reinforcement learning","year":2020,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Function (biology); Q-learning; Reinforcement; Bellman equation; Temporal difference learning; Variance (accounting); Value (mathematics)","score_opus":0.018578035697775906,"score_gpt":0.24307545516194132,"score_spread":0.2244974194641654,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6987579751","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015622927,0.00032207632,0.97977763,0.00049559295,0.00006736591,0.00004643743,0.000038551512,0.0002527524,0.0033766704],"genre_scores_gemma":[0.85059726,0.00070223404,0.14103352,0.00022251274,0.000089174195,0.00022545921,0.00010019348,0.00009849518,0.006931284],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990326,0.00043520695,0.0000490522,0.00019232939,0.00020123717,0.000089568915],"domain_scores_gemma":[0.9958627,0.0030193499,0.00030494994,0.00025786058,0.000326684,0.0002284052],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019454244,0.00084239786,0.0008606276,0.0004520167,0.00045763145,0.0012512644,0.001179411,0.0012230083,0.0037204465],"category_scores_gemma":[0.0117331715,0.00033114254,0.0005505972,0.0005490355,0.0019036677,0.0021656374,0.0013679933,0.0024981212,0.0003591065],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012378521,0.00012265897,0.0010781789,0.00010814587,0.000046358844,0.00011166416,0.00017275153,0.70679784,0.0015016679,0.22150524,0.0016515906,0.06678014],"study_design_scores_gemma":[0.00001867664,0.000044475557,0.0000866346,0.000013033014,0.000006819774,0.000013060124,0.000010132548,0.91370714,0.00045515678,0.08476455,0.00087141513,0.000008813398],"about_ca_topic_score_codex":0.0033540735,"about_ca_topic_score_gemma":0.0024322106,"teacher_disagreement_score":0.0037204465,"about_ca_system_score_codex":0.0014399608,"about_ca_system_score_gemma":0.0014117224,"threshold_uncertainty_score":0.0124461055},"labels":[],"label_agreement":null},{"id":"W6991803794","doi":"","title":"Inductive biases and generalisation for deep reinforcement learning","year":2021,"lang":"en","type":"dissertation","venue":"Oxford University Research Archive (ORA) (University of Oxford)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kellogg's (Canada)","funders":"","keywords":"Reinforcement learning; Focus (optics); Scope (computer science); Transfer of learning; Artificial neural network; Deep learning; Moment (physics); Inductive bias","score_opus":0.04855249049508199,"score_gpt":0.28390067756952664,"score_spread":0.23534818707444466,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6991803794","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0061101955,0.0006492441,0.9876453,0.0006025795,0.00008953675,0.00005393357,0.00005681443,0.0005505065,0.004241834],"genre_scores_gemma":[0.5577418,0.0016484235,0.4212341,0.0012772455,0.00044467728,0.00060723134,0.0004096395,0.00064121705,0.015995583],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9989981,0.00038223987,0.000053661213,0.00023978179,0.0002359358,0.00009021975],"domain_scores_gemma":[0.9972518,0.0018208108,0.0001796985,0.00035147875,0.00030850503,0.00008762943],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024260622,0.0012271485,0.0009659193,0.0005910562,0.00035016375,0.0011566911,0.0017791875,0.0015886455,0.004984211],"category_scores_gemma":[0.010275049,0.00062136794,0.0010619704,0.000583298,0.0018769124,0.0024776538,0.002587567,0.0046732267,0.0010074333],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006014316,0.000048537415,0.00051298813,0.00017063384,0.00007235503,0.000057479545,0.00011421844,0.7700201,0.0023137021,0.11745827,0.0030679968,0.10610342],"study_design_scores_gemma":[0.000010370685,0.00003297271,0.000068737485,0.000021782307,0.000008233239,0.000013506085,0.000005720953,0.9204228,0.00068588683,0.07667666,0.0020453641,0.000008034774],"about_ca_topic_score_codex":0.00300576,"about_ca_topic_score_gemma":0.0030464444,"teacher_disagreement_score":0.004984211,"about_ca_system_score_codex":0.001976695,"about_ca_system_score_gemma":0.0009236544,"threshold_uncertainty_score":0.016673803},"labels":[],"label_agreement":null},{"id":"W7000794688","doi":"","title":"Hard attention finding using reinforcement learning","year":2025,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Reinforcement; Control (management); Action (physics); Matching (statistics)","score_opus":0.034519857057459157,"score_gpt":0.27209616070439535,"score_spread":0.2375763036469362,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7000794688","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15037455,0.0010996608,0.82249224,0.0016557914,0.00083513616,0.00022446077,0.000092070564,0.0044406825,0.018785378],"genre_scores_gemma":[0.8743337,0.00021806358,0.108786374,0.0002842014,0.00016125468,0.000112359085,0.00011456521,0.00024083725,0.015748627],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99953926,0.0001066511,0.000018626883,0.00016079849,0.000103002145,0.00007177486],"domain_scores_gemma":[0.9981394,0.0010308897,0.00010730961,0.00017740633,0.00033397035,0.00021098892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00077284785,0.0007515043,0.0009194967,0.00046103177,0.0005327996,0.0010472484,0.001310685,0.0007809113,0.008251762],"category_scores_gemma":[0.004947634,0.00032393856,0.00040615394,0.00028951588,0.0005649288,0.0011871792,0.0015311276,0.001957943,0.001304977],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011271964,0.0008901428,0.0027805811,0.00017286083,0.0001353502,0.00014822277,0.00016050626,0.21435815,0.026692847,0.011524897,0.019498114,0.7225112],"study_design_scores_gemma":[0.000058815265,0.00009753266,0.00045016813,0.000008604105,0.000021568467,0.000026238822,0.000020407588,0.9875844,0.004339749,0.0062887776,0.001093192,0.0000104247165],"about_ca_topic_score_codex":0.0053258264,"about_ca_topic_score_gemma":0.0046963166,"teacher_disagreement_score":0.008251762,"about_ca_system_score_codex":0.0008130966,"about_ca_system_score_gemma":0.00082496466,"threshold_uncertainty_score":0.027604878},"labels":[],"label_agreement":null},{"id":"W7006393979","doi":"","title":"Towards building model-based reinforcement learning agents that effectively adapt and generalize","year":2025,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Control (management); Key (lock); Action (physics); Feature (linguistics); Stability (learning theory)","score_opus":0.029598980048585154,"score_gpt":0.268028614514628,"score_spread":0.23842963446604284,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7006393979","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028954078,0.00009616094,0.9660364,0.00028554082,0.000051604835,0.00006524751,0.000032663367,0.0015127065,0.0029655523],"genre_scores_gemma":[0.5208539,0.00016834721,0.47213584,0.0002546838,0.000027820384,0.00024306515,0.00016029019,0.0002531031,0.005902933],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997733,0.00006191001,0.000014423193,0.000061045066,0.000058404075,0.000030913863],"domain_scores_gemma":[0.9993961,0.00027238464,0.000057687175,0.00010923627,0.00011446967,0.00005018621],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00067017134,0.00056407973,0.00058760314,0.00028878756,0.00036712922,0.00085513643,0.0010283941,0.00092814205,0.0022350152],"category_scores_gemma":[0.002490688,0.000495969,0.0005905987,0.00017674784,0.00066293654,0.0010427744,0.001660725,0.0015521307,0.0008266813],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000082920495,0.00018053972,0.0011114082,0.00007112249,0.000055613036,0.00008013984,0.00012301246,0.8614561,0.012329688,0.014077355,0.0024294297,0.10800265],"study_design_scores_gemma":[0.000011479903,0.000021947244,0.000043719672,0.0000033625524,0.0000064136248,0.0000098024675,0.0000071908566,0.99449795,0.0014698922,0.0032485968,0.0006767994,0.000002837741],"about_ca_topic_score_codex":0.003244981,"about_ca_topic_score_gemma":0.0042537367,"teacher_disagreement_score":0.003244981,"about_ca_system_score_codex":0.00050293107,"about_ca_system_score_gemma":0.0009011041,"threshold_uncertainty_score":0.007476926},"labels":[],"label_agreement":null},{"id":"W7008828355","doi":"","title":"Continuous coordination as a realistic scenario for lifelong learning","year":2021,"lang":"en","type":"other","venue":"Open MIND","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Université de Montréal","keywords":"Forge; Context (archaeology); ESPACE","score_opus":0.040751536259580244,"score_gpt":0.32624474546161075,"score_spread":0.28549320920203053,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7008828355","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6744412,0.00066448114,0.29990652,0.0012106983,0.00022454778,0.0003572727,0.00082205207,0.0010756103,0.021297608],"genre_scores_gemma":[0.9785755,0.00008481602,0.016762227,0.0000627758,0.000013576167,0.00015995621,0.0002698866,0.0000310678,0.0040401705],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991823,0.00034423027,0.000029615812,0.00016674465,0.0001293483,0.00014774842],"domain_scores_gemma":[0.99825126,0.00088312966,0.00010969249,0.00023037186,0.00022020047,0.00030531918],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009959827,0.00046245588,0.00053518155,0.00029336388,0.00055714726,0.0012110677,0.0011481544,0.0015899605,0.005343142],"category_scores_gemma":[0.0037406904,0.00023204322,0.00037196104,0.00028926262,0.0007705615,0.001096189,0.0010420816,0.0010500606,0.0007164908],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004510374,0.0002471858,0.0027946453,0.00012362916,0.000045335557,0.00042847692,0.00020928831,0.9619862,0.002952449,0.011517435,0.0019391711,0.017305208],"study_design_scores_gemma":[0.0000665158,0.00028092175,0.0014709712,0.00001835057,0.000010896737,0.00007696338,0.0001473681,0.98620635,0.001543817,0.007489054,0.0026680937,0.000020714573],"about_ca_topic_score_codex":0.0059268395,"about_ca_topic_score_gemma":0.0056213397,"teacher_disagreement_score":0.0059268395,"about_ca_system_score_codex":0.0009685285,"about_ca_system_score_gemma":0.0007647816,"threshold_uncertainty_score":0.017874599},"labels":[],"label_agreement":null},{"id":"W7015310510","doi":"","title":"A Study of Augmentation-Sensitivity in Reinforcement Learning","year":2025,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Reinforcement; Control (management); Action (physics); Stability (learning theory)","score_opus":0.02194920132953579,"score_gpt":0.26966483510707484,"score_spread":0.24771563377753905,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7015310510","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3209641,0.002119686,0.6342963,0.0040460993,0.00018212473,0.00009175396,0.00008630196,0.00044702788,0.037766658],"genre_scores_gemma":[0.98116654,0.00034153525,0.015275516,0.00015583646,0.000073208685,0.00003792858,0.000017065244,0.00005537978,0.0028770082],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980531,0.0011647602,0.00008215931,0.00026968922,0.0002734729,0.00015670848],"domain_scores_gemma":[0.9052645,0.08869249,0.0019180106,0.0014899139,0.0016158157,0.0010192969],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004354511,0.00062922825,0.0011634097,0.0006384599,0.0005708296,0.0019023563,0.001372127,0.0012212348,0.004593015],"category_scores_gemma":[0.054053944,0.00074899907,0.0010561765,0.0007390516,0.002954846,0.003751794,0.0018024755,0.003496673,0.0001470203],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036879326,0.00017662207,0.0034362765,0.00024507433,0.00016309062,0.00034786854,0.00063200743,0.3232863,0.004297874,0.62191343,0.001833727,0.04329897],"study_design_scores_gemma":[0.00003012675,0.00010662684,0.000650236,0.000024106459,0.00003535899,0.00008362231,0.000045043278,0.7651606,0.0008297104,0.23250178,0.0005118537,0.000020973075],"about_ca_topic_score_codex":0.0028929836,"about_ca_topic_score_gemma":0.0010076823,"teacher_disagreement_score":0.004593015,"about_ca_system_score_codex":0.0014692156,"about_ca_system_score_gemma":0.00091871124,"threshold_uncertainty_score":0.023029089},"labels":[],"label_agreement":null},{"id":"W7015655229","doi":"","title":"Towards alignment of Reinforcement Learning agents; for consideration of safety, robustness and fairness.","year":2024,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Robustness (evolution); Reinforcement learning; Control theory (sociology); Control (management)","score_opus":0.02355615140994175,"score_gpt":0.2650591302153393,"score_spread":0.24150297880539753,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7015655229","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010212411,0.0011591997,0.9570061,0.0018691639,0.00024464817,0.00007260001,0.000054675707,0.00027879653,0.02910244],"genre_scores_gemma":[0.51265854,0.0014794938,0.4473557,0.00087621587,0.00042175743,0.00035519365,0.00025014672,0.00046398232,0.036138967],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.999042,0.00040651875,0.000040867006,0.0001975,0.00022386172,0.00008924012],"domain_scores_gemma":[0.998268,0.00090796023,0.00021624527,0.00022066032,0.00019411332,0.00019300247],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020150382,0.0006547939,0.0006349349,0.00034024668,0.0006172977,0.0012636568,0.0010444695,0.0011307725,0.008618779],"category_scores_gemma":[0.009103605,0.00040313668,0.000652857,0.00043841862,0.0018111418,0.0023180698,0.0028477937,0.0031708283,0.0013761282],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008896382,0.00005627678,0.00055648707,0.00010300731,0.000047842586,0.00006520363,0.00029134942,0.11616324,0.0015609997,0.80729187,0.007751024,0.066023625],"study_design_scores_gemma":[0.000033143104,0.00005620916,0.00024585242,0.0000519219,0.000015801961,0.000034201636,0.00005666779,0.35515073,0.00094626175,0.6243763,0.019019943,0.000012951811],"about_ca_topic_score_codex":0.0011353272,"about_ca_topic_score_gemma":0.0009652679,"teacher_disagreement_score":0.008618779,"about_ca_system_score_codex":0.0011919116,"about_ca_system_score_gemma":0.0012107902,"threshold_uncertainty_score":0.028832674},"labels":[],"label_agreement":null},{"id":"W7015986514","doi":"","title":"Towards better generalization capabilities of reinforcement learning agents via self-supervision","year":2023,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Generalization; Reinforcement learning; Control (management); Feature (linguistics); Stability (learning theory)","score_opus":0.019249336379743232,"score_gpt":0.25199970022537843,"score_spread":0.2327503638456352,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7015986514","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07170885,0.00013136797,0.9251217,0.00040065937,0.000022654218,0.000062674626,0.000023504868,0.00077562965,0.0017529038],"genre_scores_gemma":[0.88057023,0.00008998442,0.117292315,0.00016848653,0.000026605992,0.00010842564,0.000051218994,0.000072236144,0.001620534],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991912,0.00031691964,0.000047920043,0.00019474197,0.00016201899,0.00008725786],"domain_scores_gemma":[0.9952538,0.0026027297,0.000473373,0.0009973111,0.0004741657,0.00019856582],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030422185,0.0008834894,0.0009760351,0.0003474503,0.00033592834,0.0007967641,0.0013745308,0.0009966566,0.0011378792],"category_scores_gemma":[0.009958453,0.0005475017,0.00069185067,0.00022819491,0.0015069599,0.001983671,0.0021023124,0.0024804461,0.0002955691],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008489422,0.0000970927,0.0010523797,0.00004814094,0.000056576344,0.000053124135,0.00016566853,0.9223219,0.0042137904,0.022851517,0.0006357517,0.04841915],"study_design_scores_gemma":[0.0000061457704,0.000024342391,0.0000434273,0.0000023851849,0.0000020286325,0.0000038348494,0.00000298236,0.99484247,0.00031215636,0.004679281,0.00007882273,0.0000021233684],"about_ca_topic_score_codex":0.003379563,"about_ca_topic_score_gemma":0.0027556245,"teacher_disagreement_score":0.003379563,"about_ca_system_score_codex":0.0009292933,"about_ca_system_score_gemma":0.0009895783,"threshold_uncertainty_score":0.016089022},"labels":[],"label_agreement":null},{"id":"W7021226743","doi":"","title":"North American Energy Partners Inc. Announces Results for the Fourth Quarter Ended December 31, 2017","year":2018,"lang":"en","type":"other","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Quarter (Canadian coin); Energy (signal processing); Fourth World","score_opus":0.027513206543679274,"score_gpt":0.28858685765101116,"score_spread":0.2610736511073319,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7021226743","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0029199985,0.0011084455,0.0035389364,0.028182847,0.011065053,0.00034723777,0.013409071,0.00371836,0.9357101],"genre_scores_gemma":[0.006857354,0.00032020593,0.00043358747,0.0018510222,0.00036699537,0.00008230175,0.0042225188,0.00048205635,0.9853839],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9982237,0.000109173074,0.000027523,0.00011003375,0.001155844,0.00037379822],"domain_scores_gemma":[0.9960211,0.0004052263,0.00008951574,0.00024739318,0.0023865653,0.0008502207],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0025049138,0.00084563857,0.00072274165,0.0012917146,0.0025832301,0.006965299,0.0012566749,0.0033681842,0.41545856],"category_scores_gemma":[0.0055557834,0.00039126852,0.00062360434,0.0010935523,0.0007769994,0.0029521736,0.0022788327,0.0032563847,0.22520077],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006417349,0.0000743523,0.0001669766,0.000022542808,0.0000025828344,0.000019887679,0.000008331197,0.00009417207,0.000115730494,0.0026081998,0.9842053,0.012617827],"study_design_scores_gemma":[0.000029228608,0.000051882293,0.000958786,0.000033428958,0.0000046632877,0.000016033517,0.00009715127,0.00057970005,0.00061962206,0.0025717446,0.99502605,0.000011714161],"about_ca_topic_score_codex":0.017740013,"about_ca_topic_score_gemma":0.055063784,"teacher_disagreement_score":0.41545856,"about_ca_system_score_codex":0.002578275,"about_ca_system_score_gemma":0.0050005307,"threshold_uncertainty_score":0.8337774},"labels":[],"label_agreement":null},{"id":"W7023516600","doi":"","title":"Ontario's vaccine rollout plan and charges in Beirut blast: In The News for Dec. 11","year":2020,"lang":"en","type":"other","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Plan (archaeology); Government (linguistics); Action plan; Work (physics)","score_opus":0.02289565758732529,"score_gpt":0.2388574012022127,"score_spread":0.2159617436148874,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7023516600","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032155758,0.0049124393,0.000778559,0.38669828,0.019560523,0.0002651018,0.017354457,0.001143548,0.5660715],"genre_scores_gemma":[0.019451179,0.0019425284,0.00070746976,0.0718415,0.0011443748,0.00008788232,0.005105484,0.00028110892,0.8994385],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99783236,0.0001039469,0.00004498311,0.0000769973,0.0011819522,0.0007598412],"domain_scores_gemma":[0.9973767,0.0002003497,0.00007463894,0.00004720771,0.001365518,0.00093559705],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015516948,0.0006030601,0.0004853859,0.0006097883,0.004769986,0.003183585,0.0011178332,0.008315499,0.0826563],"category_scores_gemma":[0.0053522293,0.0004805207,0.00054785644,0.0006890207,0.0011508063,0.0011469219,0.0014491058,0.0054234443,0.026432808],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000023313576,0.000005833855,0.00021709279,0.000012041913,0.0000018101367,0.000032720207,0.000019988569,0.000030300025,0.000040303807,0.00090704503,0.99591404,0.0027954532],"study_design_scores_gemma":[0.00002410537,0.0000141972605,0.0025985765,0.000040562045,0.0000062793742,0.000018996512,0.00015463133,0.00010516259,0.00011067604,0.00038764824,0.99652594,0.00001325681],"about_ca_topic_score_codex":0.8978443,"about_ca_topic_score_gemma":0.97688055,"teacher_disagreement_score":0.102155685,"about_ca_system_score_codex":0.021654444,"about_ca_system_score_gemma":0.05828691,"threshold_uncertainty_score":0.27651286},"labels":[],"label_agreement":null},{"id":"W7023517986","doi":"","title":"O’Reilly Automotive, Inc. Reports Fourth Quarter and Full-Year 2017 Results and Announces Additional $1.0 Billion Share Repurchase Authorization","year":2018,"lang":"en","type":"other","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Quarter (Canadian coin); Authorization; Payment; Government (linguistics)","score_opus":0.01473504582704913,"score_gpt":0.24412804123024037,"score_spread":0.22939299540319125,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7023517986","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0016861324,0.00044050315,0.0027335829,0.0035341945,0.002391345,0.00024896214,0.0053972476,0.0048329914,0.978735],"genre_scores_gemma":[0.004644887,0.00018207704,0.00032390465,0.0003836802,0.00016843775,0.000043120035,0.0034468803,0.00057635206,0.9902305],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99707603,0.00009515525,0.00004887981,0.00018643646,0.0021578975,0.0004355779],"domain_scores_gemma":[0.9935189,0.00041751223,0.00014691251,0.00045176092,0.0041470523,0.0013179193],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.002061974,0.001242641,0.00069314486,0.002218315,0.0028213407,0.0062775426,0.0014984402,0.002923924,0.60953224],"category_scores_gemma":[0.005605495,0.0006091993,0.0007299933,0.001227228,0.00084232684,0.0035888643,0.0019699184,0.0025848537,0.48354676],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000054801647,0.00009533163,0.0002552981,0.00003222741,0.0000030370036,0.000034177112,0.000017736702,0.000081107064,0.00048519473,0.0019477283,0.9654475,0.031545907],"study_design_scores_gemma":[0.000031544605,0.000058559504,0.0007658288,0.000028897059,0.000005299952,0.000039850343,0.00008927521,0.00073159084,0.0014058088,0.0012482941,0.9955782,0.000016744716],"about_ca_topic_score_codex":0.032217205,"about_ca_topic_score_gemma":0.08595497,"teacher_disagreement_score":0.39046776,"about_ca_system_score_codex":0.0023468046,"about_ca_system_score_gemma":0.0047167772,"threshold_uncertainty_score":0.55695486},"labels":[],"label_agreement":null},{"id":"W7024241891","doi":"","title":"Responsible, Sustainable Production and Consumption through the lens of Inclusion and Diversity.","year":2025,"lang":"en","type":"other","venue":"OCAD University Open Research Repository (OCAD University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Sustainability; Inclusion (mineral); Consumption (sociology); Production (economics); Sustainable consumption; Immigration; Sustainable development; Product (mathematics); Social exclusion","score_opus":0.05194559815452137,"score_gpt":0.2956143282971589,"score_spread":0.2436687301426375,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7024241891","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.052993942,0.01856056,0.023865439,0.08889948,0.00059852086,0.000053651853,0.000095461786,0.00012392754,0.814809],"genre_scores_gemma":[0.93900156,0.006081205,0.008206394,0.002459623,0.00016965476,0.00007645145,0.000033657903,0.00007949905,0.04389185],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9975363,0.0017945576,0.000039253333,0.00017575538,0.0002724036,0.00018172088],"domain_scores_gemma":[0.99860126,0.000731139,0.00012620974,0.0001469588,0.000117386444,0.0002770899],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026334117,0.00060069695,0.00025300775,0.00160336,0.006533654,0.009968759,0.0007820655,0.0022963774,0.0053336928],"category_scores_gemma":[0.0022664357,0.00016656493,0.0002791971,0.0012743257,0.035915196,0.0076609934,0.0072913934,0.0025083404,0.0005742728],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000123342525,0.000019964567,0.00060738716,0.00010698004,0.000005310525,0.000134832,0.05716212,0.00026974638,0.00015808786,0.9072313,0.0075784833,0.026713436],"study_design_scores_gemma":[0.0000071483737,0.000024751012,0.0017098936,0.00045183185,0.000007792954,0.00020891924,0.061595723,0.0005186531,0.00044012352,0.5608687,0.37415016,0.000016239788],"about_ca_topic_score_codex":0.01719811,"about_ca_topic_score_gemma":0.03203006,"teacher_disagreement_score":0.01719811,"about_ca_system_score_codex":0.008147886,"about_ca_system_score_gemma":0.0052973223,"threshold_uncertainty_score":0.059117258},"labels":[],"label_agreement":null},{"id":"W7024873855","doi":"","title":"Summary of the Functional Foods and Nutraceuticals Survey","year":2017,"lang":"en","type":"article","venue":"AgEcon Search (University of Minnesota, USA)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Nutraceutical; Baseline (sea); Food sector; Agriculture; Intellectual property; Functional food","score_opus":0.07801678161677084,"score_gpt":0.27041173991398765,"score_spread":0.1923949582972168,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7024873855","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012959194,0.0011655836,0.00053879595,0.0019737151,0.00023056577,0.0013584292,0.9591297,0.00019859604,0.02244548],"genre_scores_gemma":[0.037195995,0.0053486363,0.0027141757,0.0017520469,0.00023749023,0.0037483573,0.9150828,0.00013404858,0.033786546],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.9962747,0.0003160473,0.00040488964,0.00031026418,0.0019772172,0.00071686355],"domain_scores_gemma":[0.98294985,0.0006498518,0.0006986282,0.00026304467,0.014443885,0.000994709],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032960419,0.00075391337,0.00081482535,0.006933827,0.001275972,0.0017220622,0.0014156152,0.0006887682,0.032883555],"category_scores_gemma":[0.009267225,0.0005041319,0.0006060754,0.012677524,0.0002886758,0.0009574262,0.0010579989,0.0010157378,0.0151487235],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017046837,0.00014740348,0.07838499,0.0011663617,0.000049613194,0.000091695445,0.0005719914,0.0003724082,0.00024192422,0.0011954682,0.8693283,0.048279468],"study_design_scores_gemma":[0.00003695459,0.00009106156,0.46399966,0.00067574414,0.000030338271,0.000101845326,0.001826041,0.00029263084,0.00021342801,0.00022805059,0.53244776,0.000056484714],"about_ca_topic_score_codex":0.7656188,"about_ca_topic_score_gemma":0.7444484,"teacher_disagreement_score":0.7656188,"about_ca_system_score_codex":0.013369834,"about_ca_system_score_gemma":0.023130247,"threshold_uncertainty_score":0.471523},"labels":[],"label_agreement":null},{"id":"W7030246915","doi":"","title":"Network architecture enforced symmetry","year":2019,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Gait; Variety (cybernetics); Symmetry (geometry); Work (physics); Task (project management); Space (punctuation); Motion (physics); Action (physics)","score_opus":0.012529671805740791,"score_gpt":0.23157169526475793,"score_spread":0.21904202345901713,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7030246915","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21650472,0.0006285128,0.7359071,0.0010376328,0.00045371454,0.0002580351,0.0007623003,0.003361275,0.041086756],"genre_scores_gemma":[0.9334535,0.00017987155,0.055954777,0.00027707603,0.000037447197,0.00017790997,0.00065273925,0.00015381533,0.009112792],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999522,0.00008304395,0.000028544127,0.000138176,0.0001248984,0.000103310354],"domain_scores_gemma":[0.9989102,0.00033756904,0.00012735453,0.00020480026,0.00033578958,0.00008438188],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007963603,0.0009842649,0.0006788869,0.00040218318,0.00042970935,0.00077273813,0.0010387319,0.00083165936,0.00682592],"category_scores_gemma":[0.0036774778,0.00027366113,0.0005868216,0.00023644287,0.0005670935,0.0011120453,0.0010563703,0.0013710564,0.0010479375],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022790449,0.00011962958,0.0016908245,0.00012917486,0.000054070093,0.00018273857,0.000044255983,0.8536067,0.009971101,0.013609978,0.0045319097,0.11583168],"study_design_scores_gemma":[0.00001362374,0.00007235469,0.00024945458,0.000011283087,0.000009638161,0.000034766425,0.000007281998,0.99265105,0.0026157517,0.0036184504,0.00071149867,0.000004862315],"about_ca_topic_score_codex":0.005280257,"about_ca_topic_score_gemma":0.005406744,"teacher_disagreement_score":0.00682592,"about_ca_system_score_codex":0.0011870611,"about_ca_system_score_gemma":0.0014014617,"threshold_uncertainty_score":0.022835016},"labels":[],"label_agreement":null},{"id":"W7033768098","doi":"","title":"Sand Filling Activities-Salinity Monitoring Matang Federal Administrative Centre (FAC) Access Road, Kuching Division, Sarawak (2rd Quarter 2004)","year":2004,"lang":"en","type":"report","venue":"Unimas Institutional Repository (Universiti Malaysia Sarawak)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Quarter (Canadian coin); Work (physics)","score_opus":0.04250607032389964,"score_gpt":0.29324271924765444,"score_spread":0.2507366489237548,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7033768098","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6605349,0.00044019675,0.016031878,0.0033161144,0.00065731024,0.0025873403,0.097642586,0.0059724506,0.21281713],"genre_scores_gemma":[0.7396216,0.0006163468,0.022511277,0.00023707644,0.000047844715,0.00044463875,0.045600776,0.00024251253,0.19067799],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99891686,0.000100635865,0.0000725133,0.00017551826,0.0005037714,0.00023063013],"domain_scores_gemma":[0.99661046,0.00021935558,0.0002596509,0.0003507083,0.0020379429,0.0005217412],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006855482,0.0006879367,0.00039844488,0.0011857711,0.0011594925,0.0011060159,0.0010455546,0.0007769972,0.013608413],"category_scores_gemma":[0.002168343,0.000514934,0.00028775499,0.0018690572,0.00039899716,0.0006874291,0.0009981861,0.00059175264,0.008558747],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020248706,0.0012741077,0.40254393,0.00038886192,0.00011831467,0.00061717327,0.0020502745,0.012613445,0.015603834,0.000986401,0.18721694,0.37456185],"study_design_scores_gemma":[0.000088238994,0.0008415493,0.856373,0.000088022985,0.0000712022,0.00030498896,0.0032760713,0.016584883,0.029824037,0.00047850082,0.09199609,0.0000734816],"about_ca_topic_score_codex":0.3730565,"about_ca_topic_score_gemma":0.5522157,"teacher_disagreement_score":0.3730565,"about_ca_system_score_codex":0.004606475,"about_ca_system_score_gemma":0.008778797,"threshold_uncertainty_score":0.74177015},"labels":[],"label_agreement":null},{"id":"W7033872038","doi":"","title":"Santander Consumer USA Holdings Inc. Reports First Quarter 2017 Results","year":2017,"lang":"en","type":"other","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Quarter (Canadian coin); Consumption (sociology); Consumer behaviour; Work (physics)","score_opus":0.02679481426724293,"score_gpt":0.2738101485653214,"score_spread":0.24701533429807845,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7033872038","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0036584816,0.0020063983,0.011006791,0.007013173,0.0032855053,0.0005262386,0.014610841,0.0056938035,0.95219886],"genre_scores_gemma":[0.020511514,0.0025680752,0.00544741,0.0013090341,0.00047711103,0.0002604851,0.020797046,0.0018672795,0.94676214],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99713194,0.00025674546,0.000100674035,0.00025772478,0.0018766264,0.000376169],"domain_scores_gemma":[0.9922167,0.0011408613,0.00020532026,0.001362517,0.003994046,0.0010804951],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0043968908,0.0013455251,0.0009309345,0.0025836308,0.0012993198,0.0045225825,0.0017987192,0.0013660184,0.5071677],"category_scores_gemma":[0.00868997,0.00039060312,0.0014153443,0.0015629795,0.0007842828,0.0022806092,0.0021420065,0.0023035049,0.3584],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017744175,0.00054435135,0.0005274047,0.00017298649,0.00002206615,0.000038824368,0.000018308612,0.0011832467,0.0006571048,0.0058986736,0.8403222,0.15043741],"study_design_scores_gemma":[0.00010184945,0.00019530242,0.0018546802,0.00017415591,0.00003586744,0.000076988734,0.000086631204,0.0027585167,0.00439608,0.0066157123,0.983679,0.000025208617],"about_ca_topic_score_codex":0.01378298,"about_ca_topic_score_gemma":0.023185227,"teacher_disagreement_score":0.4928323,"about_ca_system_score_codex":0.0028416838,"about_ca_system_score_gemma":0.00505881,"threshold_uncertainty_score":0.7029655},"labels":[],"label_agreement":null},{"id":"W7037446629","doi":"","title":"Être parent d’un adulte atteint de schizophrénie en Mauricie–Centre-du-Québec : stratégies de coping et représentations sociales de la maladie","year":2023,"lang":"fr","type":"other","venue":"Le dépôt institutionnel (Université du Québec à Trois-Rivières)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Coping (psychology); Disease; Qualitative research; Distress","score_opus":0.01668348037815944,"score_gpt":0.2387424358042205,"score_spread":0.22205895542606105,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7037446629","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9888411,0.0011739725,0.00015424067,0.0062171915,0.00009497535,0.000028162834,0.00012390358,0.0000060933353,0.003360309],"genre_scores_gemma":[0.98528844,0.001402486,0.00036338362,0.0016658801,0.000025119254,0.000038461952,0.00010369191,0.000008261865,0.011104209],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.9997534,0.00006162944,0.000009515501,0.000027329594,0.00004184158,0.000106224135],"domain_scores_gemma":[0.99925643,0.00011458474,0.00013301623,0.000018813143,0.0001877349,0.00028932616],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00052487315,0.00036558302,0.00033386357,0.00033265472,0.004654721,0.0011376624,0.00082417304,0.0014510024,0.0035245153],"category_scores_gemma":[0.0018414318,0.00029312933,0.00028353903,0.00044674551,0.0009846138,0.0008937666,0.00071890414,0.0017921355,0.00030255658],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006566713,0.0013772227,0.50580114,0.00019289444,0.000104202154,0.044929534,0.34429064,0.0003071981,0.00431544,0.0023544398,0.018586533,0.077084005],"study_design_scores_gemma":[0.00005595669,0.00049072417,0.6424615,0.0003429612,0.00008423494,0.009413202,0.3257136,0.00059060776,0.00066942174,0.00043647434,0.019656029,0.00008538448],"about_ca_topic_score_codex":0.8173695,"about_ca_topic_score_gemma":0.9001904,"teacher_disagreement_score":0.18263048,"about_ca_system_score_codex":0.008546779,"about_ca_system_score_gemma":0.006671288,"threshold_uncertainty_score":0.36741203},"labels":[],"label_agreement":null},{"id":"W7037450537","doi":"","title":"The effects of cyclic mechanical stretch on nuclear protein import in vascular smooth muscle cells","year":2005,"lang":"en","type":"dissertation","venue":"Mspace (University of Manitoba)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Canadian Institutes of Health Research","keywords":"Vascular smooth muscle; Myocyte; Smooth muscle; Nuclear protein; Cell nucleus; Blood vessel","score_opus":0.005845724919362133,"score_gpt":0.18896707761745965,"score_spread":0.1831213526980975,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7037450537","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99763954,0.0008869994,0.00023819685,0.00018951016,0.00002291865,0.000004289924,0.00010104028,0.000012837686,0.00090470567],"genre_scores_gemma":[0.9985115,0.00048859353,0.00020891415,0.000041234733,0.000007603184,0.000006703569,0.00007852727,0.00000753313,0.0006492899],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9999075,0.000016938588,0.0000074086274,0.000014197507,0.000022968585,0.00003101523],"domain_scores_gemma":[0.999681,0.000137304,0.00004321233,0.000043830958,0.000027364646,0.00006727508],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00014806715,0.00013643269,0.00020881937,0.00013534051,0.00022428953,0.00038237448,0.00023060718,0.00030796407,0.0019486396],"category_scores_gemma":[0.00045352368,0.00014970444,0.00025156426,0.0002307737,0.0005048641,0.00028402792,0.00023465542,0.00045749298,0.00012636837],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0029058873,0.00015031162,0.0007818263,0.000094430216,0.000021538137,0.000118546894,0.00011473075,0.0012255552,0.987599,0.00031048493,0.0002170039,0.006460641],"study_design_scores_gemma":[0.0004472206,0.003208781,0.0891266,0.00004340267,0.000080835234,0.00029208272,0.0008393467,0.014151246,0.8886656,0.0007284308,0.0023696695,0.000046734258],"about_ca_topic_score_codex":0.0070530176,"about_ca_topic_score_gemma":0.011747689,"teacher_disagreement_score":0.0070530176,"about_ca_system_score_codex":0.0008047543,"about_ca_system_score_gemma":0.00040397764,"threshold_uncertainty_score":0.0140239},"labels":[],"label_agreement":null},{"id":"W7042992782","doi":"","title":"Reproducibility and reusability in deep reinforcement learning","year":2018,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reproducibility; Reinforcement learning; Reusability; Reinforcement; Reliability (semiconductor)","score_opus":0.018471971676706552,"score_gpt":0.2572029638222421,"score_spread":0.23873099214553556,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7042992782","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.083278865,0.00036602982,0.91150033,0.0005273389,0.000058203306,0.00016466726,0.00009968381,0.001331881,0.002672956],"genre_scores_gemma":[0.87540984,0.00018085605,0.121328235,0.00022120624,0.000051434417,0.00048799685,0.00021933978,0.000513823,0.0015872103],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9814802,0.009276077,0.001323345,0.0038153303,0.0033664072,0.00073879265],"domain_scores_gemma":[0.8812489,0.08211005,0.0067055104,0.024827447,0.003742258,0.0013658084],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.023063045,0.001418957,0.0016739705,0.00097004214,0.00095519895,0.0028092328,0.0037358194,0.0021432156,0.0025067672],"category_scores_gemma":[0.11939593,0.0010987216,0.001626052,0.0007708921,0.0062637255,0.00562403,0.006198639,0.0048963944,0.000535745],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005537378,0.0002853024,0.004589696,0.0003670986,0.0002037207,0.0002709039,0.00058181206,0.83320916,0.0060381293,0.081369445,0.0008012469,0.071729705],"study_design_scores_gemma":[0.0000670194,0.00027116077,0.00061711465,0.000045368703,0.000031175594,0.000059719125,0.000048021695,0.89992714,0.005065293,0.092914954,0.0009147742,0.000038238795],"about_ca_topic_score_codex":0.0020234827,"about_ca_topic_score_gemma":0.0014109436,"teacher_disagreement_score":0.97693694,"about_ca_system_score_codex":0.0021703867,"about_ca_system_score_gemma":0.0019428428,"threshold_uncertainty_score":0.121970534},"labels":[],"label_agreement":null},{"id":"W7045640502","doi":"","title":"Artificial intelligence driven decision-making under uncertainty","year":2024,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; McGill University","keywords":"Expert system; Applications of artificial intelligence; Key (lock); Artificial neural network; Feature (linguistics)","score_opus":0.027112096147988998,"score_gpt":0.29207114915231774,"score_spread":0.26495905300432876,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7045640502","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07657793,0.0009238546,0.8996386,0.0032154322,0.00016987782,0.00015884764,0.0002583141,0.00031688449,0.018740302],"genre_scores_gemma":[0.8728042,0.0011956554,0.11936703,0.00042956084,0.00015812309,0.0003123979,0.00025313534,0.00006684979,0.0054129744],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975521,0.0010357276,0.00013997438,0.00047845265,0.0004986518,0.00029520737],"domain_scores_gemma":[0.9878598,0.010044348,0.00071189203,0.0005271301,0.00056740263,0.00028950724],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00310402,0.00086895784,0.0013112513,0.0005052174,0.0007715189,0.0028985431,0.0013975032,0.001506246,0.003967128],"category_scores_gemma":[0.012955353,0.0005299596,0.0009892107,0.0007541875,0.002445617,0.0023093864,0.0018088646,0.0030543052,0.00047135566],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007147775,0.00004496328,0.00063879933,0.00011333062,0.000051184226,0.0001120667,0.00010931971,0.8316837,0.0004765331,0.1536959,0.00091946527,0.012083269],"study_design_scores_gemma":[0.000021186092,0.000020427853,0.00010516747,0.000017016571,0.000009267,0.000017054375,0.000030483949,0.8575894,0.0002658364,0.14090154,0.0010125762,0.000009957202],"about_ca_topic_score_codex":0.0046109376,"about_ca_topic_score_gemma":0.002623162,"teacher_disagreement_score":0.0046109376,"about_ca_system_score_codex":0.0023127142,"about_ca_system_score_gemma":0.00253329,"threshold_uncertainty_score":0.016780019},"labels":[],"label_agreement":null},{"id":"W7071380059","doi":"","title":"Risk-directed exploration in reinforcement learning","year":2005,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Reinforcement learning; Entropy (arrow of time); Measure (data warehouse); Action (physics); Reinforcement; Action selection; Class (philosophy)","score_opus":0.01879923180380377,"score_gpt":0.24857631707950886,"score_spread":0.2297770852757051,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7071380059","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.045085713,0.0034720378,0.86902887,0.0026457412,0.0002500728,0.00014359843,0.00017757846,0.001158226,0.07803821],"genre_scores_gemma":[0.82309145,0.0025902928,0.11378796,0.00042549236,0.00017213839,0.00042232173,0.00024097589,0.00020346075,0.059066024],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99943,0.000352959,0.000023373803,0.00007473029,0.00007418601,0.00004473688],"domain_scores_gemma":[0.997586,0.0019837103,0.00011100329,0.00011110009,0.00011657266,0.00009172364],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016877796,0.0005523792,0.0008563175,0.00031606245,0.000239545,0.001092523,0.00077083585,0.0011126795,0.018692063],"category_scores_gemma":[0.006250837,0.00029184742,0.0002942566,0.00049800676,0.00090959965,0.0014326785,0.001110323,0.0016552174,0.0028884232],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005121973,0.00024549948,0.002231056,0.00031400778,0.00014261223,0.00014874055,0.00020311022,0.45995316,0.0012531262,0.20202716,0.016614057,0.31635526],"study_design_scores_gemma":[0.000111825895,0.0001649782,0.00041588658,0.000066586,0.000017811868,0.00003949001,0.000030459923,0.8677739,0.00035583417,0.12233238,0.008675636,0.000015226119],"about_ca_topic_score_codex":0.0026999884,"about_ca_topic_score_gemma":0.00231865,"teacher_disagreement_score":0.018692063,"about_ca_system_score_codex":0.0009980852,"about_ca_system_score_gemma":0.0009543015,"threshold_uncertainty_score":0.06253123},"labels":[],"label_agreement":null},{"id":"W7072003558","doi":"","title":"Uncertainty aware behavioral cloning using Bayesian Neural Networks","year":2019,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Consistency (knowledge bases); Probabilistic logic; Artificial neural network; Scalability; Robot; Reinforcement learning","score_opus":0.02643474342105103,"score_gpt":0.2769272709484118,"score_spread":0.2504925275273608,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7072003558","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025089096,0.0001706075,0.97067416,0.00033412743,0.000029755425,0.00005041071,0.000062250045,0.00056117336,0.003028441],"genre_scores_gemma":[0.7705385,0.00030963746,0.22137158,0.00023042208,0.000042959207,0.000219989,0.0002374229,0.0001872268,0.006862142],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99889475,0.00041669895,0.000048316004,0.0002445886,0.00028238827,0.00011333843],"domain_scores_gemma":[0.9965867,0.0021585743,0.0004046072,0.000390262,0.00030559362,0.00015431445],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001851092,0.0007236307,0.0009526491,0.0006937687,0.00050222373,0.00094561634,0.002167169,0.0011578094,0.0030054145],"category_scores_gemma":[0.008027145,0.0008596232,0.0010137347,0.00057877344,0.0015981308,0.0023850799,0.0019948303,0.0021701823,0.00040285467],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006763832,0.000077713325,0.0010038444,0.000038497055,0.000044935005,0.00004308387,0.000094496376,0.9162373,0.0015152525,0.029726168,0.0007112976,0.05043971],"study_design_scores_gemma":[0.000003929542,0.00000834542,0.00009740417,0.000004904681,0.0000039091633,0.0000057114116,0.000004372985,0.9873355,0.00022698415,0.012136952,0.0001670256,0.0000048653296],"about_ca_topic_score_codex":0.009252693,"about_ca_topic_score_gemma":0.008405936,"teacher_disagreement_score":0.009252693,"about_ca_system_score_codex":0.0020990837,"about_ca_system_score_gemma":0.001385228,"threshold_uncertainty_score":0.018397689},"labels":[],"label_agreement":null},{"id":"W7093331850","doi":"10.1016/j.engappai.2025.112779","title":"Shaping <mml:math xmlns:mml=\"http://www.w3.org/1998/Math/MathML\" altimg=\"si159.svg\" display=\"inline\" id=\"d1e703\"> <mml:mi>Q</mml:mi> </mml:math> -values right: A distributional normalized actor–critic approach","year":2025,"lang":"en","type":"article","venue":"Engineering Applications of Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Institute for Information and Communications Technology Promotion; National Fire Agency; Busan Metropolitan City; Information Technology Research Centre; Korea Institute for Advancement of Technology; Ministry of Science and ICT, South Korea; Ministry of Trade, Industry and Energy; Ministry of Education","keywords":"Reinforcement learning; Key (lock); Upper and lower bounds; Stability (learning theory); Range (aeronautics); Value (mathematics); Q-learning","score_opus":0.022669345411759204,"score_gpt":0.2668609660288699,"score_spread":0.2441916206171107,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7093331850","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013026336,0.000130402,0.82142097,0.0020284732,0.0005409277,0.00013347555,0.0062282076,0.027262446,0.14095245],"genre_scores_gemma":[0.120791085,0.0005370078,0.5402162,0.0016710882,0.00033092705,0.0006025368,0.014553454,0.03315508,0.2881426],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991605,0.00022634752,0.000063029474,0.00020622861,0.00029000884,0.000053906566],"domain_scores_gemma":[0.9983577,0.00048142287,0.00006851913,0.0004436131,0.0005485727,0.00010011539],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011257645,0.0009125342,0.00067873753,0.00090261817,0.0005479123,0.0031495027,0.0020499318,0.0014391046,0.20782007],"category_scores_gemma":[0.0072772657,0.0006062378,0.0006010652,0.00088438805,0.00096592336,0.003571461,0.002432074,0.0022126588,0.10268165],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002517325,0.00009252918,0.00046319465,0.00035773116,0.000029966062,0.00018281362,0.00023916675,0.023077387,0.005704495,0.34130993,0.37384605,0.25444505],"study_design_scores_gemma":[0.0000744386,0.00003394126,0.0004469452,0.00010736345,0.000014390382,0.00017688853,0.00007176496,0.20534465,0.014496538,0.24415004,0.53502625,0.000056778015],"about_ca_topic_score_codex":0.004612441,"about_ca_topic_score_gemma":0.007676829,"teacher_disagreement_score":0.20782007,"about_ca_system_score_codex":0.0012035582,"about_ca_system_score_gemma":0.0011484755,"threshold_uncertainty_score":0.6952274},"labels":[],"label_agreement":null},{"id":"W7095254530","doi":"","title":"Regularized Reinforcement Learning with Performance Guarantees","year":2014,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Reinforcement learning; Control (management); Reinforcement; Active learning (machine learning)","score_opus":0.0067977949779677894,"score_gpt":0.19196615697294636,"score_spread":0.18516836199497858,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7095254530","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03891865,0.0011291945,0.93779606,0.0015565204,0.00022778651,0.0001413656,0.00025288513,0.0018999443,0.018077705],"genre_scores_gemma":[0.925845,0.00039760952,0.062579185,0.00022602838,0.00016995372,0.00023847936,0.00028015504,0.00022447761,0.010039055],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99842453,0.0005798744,0.00006846499,0.000369432,0.00032577146,0.00023203902],"domain_scores_gemma":[0.99520314,0.0031595393,0.0003783506,0.000573548,0.00039947307,0.00028600122],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019674131,0.0013308915,0.0013249461,0.0004436562,0.00041623373,0.0011526182,0.0011939461,0.0010053861,0.0072784782],"category_scores_gemma":[0.009936024,0.0004972789,0.00046887886,0.00050189614,0.0011484087,0.0015503899,0.0018490287,0.0028065436,0.0012954194],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00057170406,0.00022040808,0.0008604311,0.00017084053,0.000075983306,0.00011199658,0.000075296,0.8403259,0.0014607268,0.07030084,0.0097738085,0.076052055],"study_design_scores_gemma":[0.00005285024,0.000070502974,0.00009401621,0.0000121580815,0.0000071806676,0.0000145692175,0.000005927744,0.96299475,0.0002760559,0.035867568,0.00059806096,0.000006337007],"about_ca_topic_score_codex":0.002593376,"about_ca_topic_score_gemma":0.002155213,"teacher_disagreement_score":0.0072784782,"about_ca_system_score_codex":0.0015378058,"about_ca_system_score_gemma":0.0016947049,"threshold_uncertainty_score":0.024348915},"labels":[],"label_agreement":null},{"id":"W7096680464","doi":"","title":"Reinforcement Learning for Factored Markov Decision Processes","year":2002,"lang":"en","type":"article","venue":"TSpace","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Markov decision process; Inference; Partially observable Markov decision process; Representation (politics); Core (optical fiber); Action (physics); State (computer science); Markov process; Q-learning","score_opus":0.03809663181555732,"score_gpt":0.3046626435827796,"score_spread":0.2665660117672223,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7096680464","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010966713,0.0005913621,0.9846295,0.00045646093,0.000054758046,0.00005118147,0.00011180873,0.00024935685,0.0028888562],"genre_scores_gemma":[0.7997829,0.0012020934,0.19117227,0.0002179748,0.00014108827,0.0004819486,0.00041034582,0.0001151541,0.006476288],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99869955,0.00061136513,0.000062623374,0.00027597044,0.00021409441,0.00013641831],"domain_scores_gemma":[0.99323845,0.005686319,0.00039301583,0.00017926554,0.00031741668,0.00018551479],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002591402,0.0011362171,0.0016082467,0.00061583717,0.0005625222,0.0011151476,0.0014041755,0.0013328055,0.0054635815],"category_scores_gemma":[0.012553582,0.000613101,0.001001449,0.0006771423,0.0019157286,0.0017886662,0.001311421,0.0026666173,0.0005055949],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000064938715,0.000030781513,0.00041536664,0.000073841584,0.000032215034,0.000048203758,0.000056636716,0.90260094,0.00018784332,0.081912965,0.0007550878,0.013821175],"study_design_scores_gemma":[0.00001579828,0.000010688336,0.000031307784,0.000005598539,0.0000033702659,0.0000046735627,0.0000030355875,0.95403135,0.000040126066,0.045611918,0.00023865396,0.0000034368597],"about_ca_topic_score_codex":0.011252463,"about_ca_topic_score_gemma":0.008777557,"teacher_disagreement_score":0.011252463,"about_ca_system_score_codex":0.0027071445,"about_ca_system_score_gemma":0.0018797141,"threshold_uncertainty_score":0.022373915},"labels":[],"label_agreement":null},{"id":"W7097981727","doi":"","title":"Author manuscript, published in &amp;quot;Advances in Neural Information Processing Systems, Canada (2008)&amp;quot; Particle Filter-based Policy Gradient in POMDPs","year":2013,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Parameterized complexity; Resampling; Particle filter; Measure (data warehouse); Markov decision process; Variance (accounting); Process (computing); Markov process; Filter (signal processing)","score_opus":0.023871227454056985,"score_gpt":0.2540971173766306,"score_spread":0.23022588992257362,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7097981727","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009946349,0.03557477,0.23589458,0.05326576,0.19370964,0.00073645357,0.01772306,0.0075080586,0.44564143],"genre_scores_gemma":[0.03469351,0.012101613,0.0641729,0.0021340437,0.005420004,0.00016917639,0.0065469947,0.0018522991,0.8729094],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991872,0.000105766114,0.00006467746,0.00028453956,0.00027019065,0.0000877399],"domain_scores_gemma":[0.9964862,0.0006885018,0.000115403614,0.00042335168,0.0019211866,0.00036538238],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001814342,0.0009865378,0.0017641204,0.0015364484,0.0012164018,0.005730611,0.0017256669,0.0020497844,0.3161965],"category_scores_gemma":[0.008700668,0.0007275486,0.0006952229,0.0027971896,0.0011162886,0.0026486623,0.0016443541,0.0015174847,0.08234867],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003105616,0.000090734124,0.0011330291,0.00060337805,0.0001460317,0.00028806896,0.00012826106,0.006387056,0.0017606514,0.027815696,0.71474427,0.24659227],"study_design_scores_gemma":[0.00007748625,0.00006775766,0.0012656933,0.00031167275,0.000055734447,0.00028022248,0.00012201731,0.022262586,0.002853888,0.017908098,0.9547449,0.000049999835],"about_ca_topic_score_codex":0.015948702,"about_ca_topic_score_gemma":0.033215284,"teacher_disagreement_score":0.3161965,"about_ca_system_score_codex":0.002748997,"about_ca_system_score_gemma":0.003662,"threshold_uncertainty_score":0.9753627},"labels":[],"label_agreement":null},{"id":"W7102328828","doi":"","title":"Advantage Shaping as Surrogate Reward Maximization: Unifying Pass@K Policy Gradients","year":2025,"lang":"","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Perspective (graphical); Simple (philosophy); Surrogate model; Verifiable secret sharing; Optimization problem","score_opus":0.06663284614697988,"score_gpt":0.2247884175724866,"score_spread":0.15815557142550674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7102328828","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00957298,0.00017265027,0.9820895,0.0004824732,0.000047586847,0.000038760256,0.000023034287,0.00033472097,0.007238176],"genre_scores_gemma":[0.7018767,0.0004520815,0.28756768,0.00055783684,0.00011064669,0.00021485084,0.00006476667,0.0005499048,0.008605552],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986131,0.00060353376,0.00005880651,0.00022871081,0.0003692718,0.00012647665],"domain_scores_gemma":[0.9969175,0.0019052493,0.00026671714,0.0005026758,0.00024413494,0.00016379944],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030143275,0.0010315197,0.0011320934,0.0005207736,0.0004908541,0.001826453,0.0014102445,0.0016993611,0.004655837],"category_scores_gemma":[0.0116554545,0.00051831803,0.0006343987,0.00041760766,0.0031561027,0.0030645938,0.0034348657,0.0028483693,0.0009550569],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014013532,0.00006545488,0.0005575706,0.00013262319,0.000041032323,0.00010173694,0.00023013819,0.38801122,0.0031458023,0.5292142,0.0022565422,0.07610342],"study_design_scores_gemma":[0.000019356075,0.00007035573,0.00008580933,0.000039233808,0.000010853447,0.00004068438,0.000019120502,0.81435114,0.001565922,0.18101765,0.002761336,0.000018588338],"about_ca_topic_score_codex":0.0010754997,"about_ca_topic_score_gemma":0.001174189,"teacher_disagreement_score":0.004655837,"about_ca_system_score_codex":0.0015528426,"about_ca_system_score_gemma":0.0019353192,"threshold_uncertainty_score":0.015941441},"labels":[],"label_agreement":null},{"id":"W7102329548","doi":"","title":"Causal Deep Q Network","year":2025,"lang":"","type":"article","venue":"ArXiv.org","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Mitacs","keywords":"Spurious relationship; Reinforcement learning; Causal model; Causal reasoning; Benchmark (surveying); Associative property; Causal structure","score_opus":0.026820504218234478,"score_gpt":0.27053520882870463,"score_spread":0.24371470461047015,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7102329548","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012659994,0.00034750663,0.98083854,0.0005562886,0.00007758108,0.000049575992,0.00017195477,0.00057631516,0.0047221775],"genre_scores_gemma":[0.7522476,0.00073914375,0.23924142,0.0005551792,0.00010426525,0.00018489081,0.00041808403,0.00011617318,0.006393292],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99949217,0.00017078097,0.000026689173,0.00014352497,0.00011235822,0.00005450399],"domain_scores_gemma":[0.9982693,0.00093370525,0.0001850144,0.00017045412,0.00032521426,0.0001162589],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012634167,0.00061894784,0.0005998739,0.0006057947,0.000498111,0.0008344631,0.0015733435,0.001061491,0.0054250853],"category_scores_gemma":[0.005691177,0.00039463787,0.00049909623,0.0006003524,0.0012308863,0.0015865307,0.0013961272,0.00141056,0.000515428],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012725226,0.00009605029,0.0026134225,0.00018560812,0.00009063624,0.00016932595,0.00009551401,0.710234,0.0027081482,0.14023642,0.0042752814,0.13916829],"study_design_scores_gemma":[0.000011161301,0.00002184239,0.0001435401,0.000012235655,0.00001279913,0.0000210695,0.000007903098,0.9357034,0.00054987276,0.061695047,0.0018152155,0.000005808747],"about_ca_topic_score_codex":0.0054433807,"about_ca_topic_score_gemma":0.0075377603,"teacher_disagreement_score":0.0054433807,"about_ca_system_score_codex":0.0010602907,"about_ca_system_score_gemma":0.0014716141,"threshold_uncertainty_score":0.01814878},"labels":[],"label_agreement":null},{"id":"W7107957261","doi":"10.71781/32566","title":"Modeling the world to reason and plan","year":2025,"lang":"en","type":"dissertation","venue":"Open MIND","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Mitacs; Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; Institut de Valorisation des Données; Canadian Institute for Advanced Research; Nvidia","keywords":"Context (archaeology); Gloom; Demise","score_opus":0.051368406102648166,"score_gpt":0.32856710235170883,"score_spread":0.2771986962490607,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7107957261","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0070220768,0.005398368,0.91373944,0.011424384,0.00047476805,0.00021874651,0.0008138611,0.0009694324,0.05993883],"genre_scores_gemma":[0.30324847,0.009194745,0.64256805,0.0020776163,0.00048960064,0.0010383574,0.0023827814,0.0004719309,0.038528524],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9982522,0.0008884825,0.00013010723,0.00038307568,0.00024791565,0.00009823679],"domain_scores_gemma":[0.99764246,0.0012736508,0.00015584208,0.0006498351,0.00015347441,0.0001247925],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018715817,0.0013355401,0.0007710812,0.001737653,0.0010335172,0.0059023695,0.0021708142,0.0021430564,0.014599348],"category_scores_gemma":[0.006594512,0.00074674306,0.0023174149,0.0016130505,0.005236864,0.009908583,0.0042339833,0.0030649668,0.0025515507],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000044413715,0.000037550773,0.0005781076,0.00033200777,0.00008442707,0.00014094137,0.0011538563,0.033113465,0.00046225463,0.9167673,0.0050225165,0.04226296],"study_design_scores_gemma":[0.0000279628,0.000034285455,0.00021886041,0.00021760355,0.00005116996,0.00010089161,0.00051702803,0.047322612,0.0004789386,0.8593558,0.0916471,0.00002762135],"about_ca_topic_score_codex":0.009133868,"about_ca_topic_score_gemma":0.008534815,"teacher_disagreement_score":0.014599348,"about_ca_system_score_codex":0.0020895982,"about_ca_system_score_gemma":0.003161983,"threshold_uncertainty_score":0.04883969},"labels":[],"label_agreement":null},{"id":"W7110060432","doi":"10.1162/isal.a.888","title":"Energy Costs and Neural Complexity Evolution in Changing Environments","year":2025,"lang":"","type":"article","venue":"ALIFE","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institut National de la Recherche Scientifique","funders":"University of Cape Town","keywords":"Artificial neural network; Energy (signal processing); Cognition; Reinforcement learning; Key (lock); Efficient energy use","score_opus":0.020688039422258175,"score_gpt":0.2517305875269306,"score_spread":0.23104254810467245,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7110060432","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9641992,0.00015716873,0.030322982,0.00048849435,0.000021332557,0.000011384005,0.000047324236,0.000057035446,0.0046949983],"genre_scores_gemma":[0.99544996,0.000052408912,0.004024878,0.00003247961,0.000003483834,0.000011103461,0.000020791233,0.0000144561845,0.0003903637],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997842,0.00007815153,0.000010872519,0.00005160277,0.000038117556,0.00003706825],"domain_scores_gemma":[0.9988734,0.00055902835,0.00025606857,0.0001014876,0.00009444055,0.000115655595],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00047623506,0.00028177176,0.0002715269,0.00034246416,0.0003910379,0.0013548802,0.0005134724,0.000670073,0.001567879],"category_scores_gemma":[0.004431863,0.00021734623,0.0003131074,0.00018834994,0.00099059,0.0015941071,0.00095113047,0.00070354925,0.00012430274],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002228944,0.00013074279,0.04082901,0.00012751076,0.00022831015,0.0005481155,0.0005686567,0.7972043,0.048797607,0.06394418,0.0008563171,0.046542373],"study_design_scores_gemma":[0.000033212564,0.00017853221,0.031552106,0.00002656222,0.0000671318,0.0002976231,0.00038181592,0.8709286,0.0066944994,0.08732685,0.0024561272,0.00005688856],"about_ca_topic_score_codex":0.0012818926,"about_ca_topic_score_gemma":0.0016479378,"teacher_disagreement_score":0.001567879,"about_ca_system_score_codex":0.0008712649,"about_ca_system_score_gemma":0.00036310876,"threshold_uncertainty_score":0.0063215494},"labels":[],"label_agreement":null},{"id":"W7113916458","doi":"","title":"A-3PO: Accelerating Asynchronous LLM Training with Staleness-aware Proximal Policy Approximation","year":2025,"lang":"","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada)","funders":"","keywords":"Asynchronous communication; Reinforcement learning; Overhead (engineering); Interpolation (computer graphics); Stability (learning theory); Speedup; Constraint (computer-aided design); Code (set theory)","score_opus":0.0748734437683085,"score_gpt":0.2104304695268579,"score_spread":0.1355570257585494,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7113916458","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007199642,0.00012826842,0.98579156,0.00019569066,0.00007877844,0.000050288687,0.00006403212,0.0042867973,0.0022049977],"genre_scores_gemma":[0.5101306,0.00015087836,0.47882438,0.00064968236,0.0001131503,0.00037272094,0.00042439887,0.0011553158,0.008178825],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99936587,0.00014326314,0.00003411825,0.0001587134,0.00018859658,0.00010950864],"domain_scores_gemma":[0.9983936,0.0008742045,0.00011523713,0.00027917285,0.00021027542,0.0001275492],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014059298,0.0011056162,0.0011612439,0.0004388558,0.0004442696,0.0011682299,0.0022033846,0.001566026,0.0078090285],"category_scores_gemma":[0.0071324552,0.00059409277,0.0006425936,0.00039827326,0.0009169399,0.0013970014,0.0022872733,0.0023933926,0.0025095423],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027548074,0.00014844257,0.00090389303,0.00015253254,0.000051093466,0.00008590673,0.00011090662,0.79532176,0.0041799517,0.01016332,0.0077225757,0.18088408],"study_design_scores_gemma":[0.0000148361005,0.000016212225,0.000031617463,0.0000047790936,0.000003290671,0.000008473169,0.0000039843635,0.9966955,0.00068447925,0.0019930324,0.00054099,0.000002880383],"about_ca_topic_score_codex":0.0060002557,"about_ca_topic_score_gemma":0.0071708057,"teacher_disagreement_score":0.0078090285,"about_ca_system_score_codex":0.0009566694,"about_ca_system_score_gemma":0.0024316413,"threshold_uncertainty_score":0.026123822},"labels":[],"label_agreement":null},{"id":"W7117850973","doi":"","title":"The World Is Bigger! A Computationally-Embedded Perspective on the Big World Hypothesis","year":2025,"lang":"","type":"article","venue":"ArXiv.org","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Perspective (graphical); Construct (python library); Markov decision process; Automaton; Process (computing); Learning automata; Variety (cybernetics)","score_opus":0.057509890548072894,"score_gpt":0.29628165420160074,"score_spread":0.23877176365352784,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7117850973","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18675712,0.0005330404,0.78865546,0.006899289,0.00007732789,0.000054529297,0.00010367489,0.00025028732,0.016669217],"genre_scores_gemma":[0.947939,0.00019223451,0.049823184,0.0002461281,0.00003653452,0.00007715233,0.00004263284,0.000048483158,0.0015947322],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99925476,0.000358316,0.000024082241,0.00016892818,0.00012403233,0.000069892325],"domain_scores_gemma":[0.9950629,0.0035021063,0.0004669829,0.0004570928,0.00020600625,0.0003049137],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001679407,0.0005430899,0.00046764285,0.00029802351,0.00059080555,0.0013244264,0.0010698981,0.0012295634,0.0026280445],"category_scores_gemma":[0.0077615725,0.00031698917,0.0004903827,0.00022151964,0.0049726367,0.004430704,0.0018643972,0.0026380522,0.00016993364],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011593998,0.00007431844,0.0014854916,0.00012855325,0.00004916142,0.0001691063,0.0002733311,0.51750505,0.002777288,0.4603931,0.0011953599,0.015833285],"study_design_scores_gemma":[0.000022586633,0.000060841125,0.00037976893,0.000025433083,0.000010621922,0.00003935548,0.00006429512,0.6575347,0.0013315958,0.3389705,0.0015414269,0.000018878103],"about_ca_topic_score_codex":0.0015220086,"about_ca_topic_score_gemma":0.0011937966,"teacher_disagreement_score":0.0026280445,"about_ca_system_score_codex":0.0009794828,"about_ca_system_score_gemma":0.0009423811,"threshold_uncertainty_score":0.0088816285},"labels":[],"label_agreement":null},{"id":"W7118190000","doi":"10.1145/3765766.3787117","title":"10.1145/3765766.3787117","year":2000,"lang":"en","type":"article","venue":"Time to knit","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Session (web analytics); Adaptation (eye); Component (thermodynamics); Key (lock)","score_opus":0.006354011713904454,"score_gpt":0.180076401831819,"score_spread":0.17372239011791454,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7118190000","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002639765,0.0080623645,0.019008316,0.0014607579,0.002297067,0.00057320367,0.018902702,0.022864513,0.9241913],"genre_scores_gemma":[0.0044523883,0.0026210372,0.0025535235,0.0006788359,0.00014147203,0.00024294438,0.008560484,0.0019367607,0.9788126],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99936384,0.000045241683,0.000054085525,0.00020336645,0.00020907774,0.00012438599],"domain_scores_gemma":[0.9986505,0.0003235053,0.00007250641,0.00045743308,0.0002801826,0.0002159011],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.001513506,0.0036320311,0.0025312714,0.002410292,0.0018591302,0.0037506856,0.002988333,0.0053889,0.93651235],"category_scores_gemma":[0.0028018209,0.00181279,0.0012565827,0.007688543,0.0013712262,0.008049776,0.0053765145,0.0029504602,0.9568113],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002749915,0.00019474003,0.00041303004,0.0006248296,0.00004606565,0.00018545258,0.000066712135,0.0010622765,0.0013692155,0.0053221094,0.6079403,0.38250023],"study_design_scores_gemma":[0.000048171827,0.000041261388,0.00074093044,0.00027525137,0.000047314763,0.00014916,0.0000519174,0.0014079583,0.00070408353,0.0018852354,0.9946136,0.000035097157],"about_ca_topic_score_codex":0.015680406,"about_ca_topic_score_gemma":0.010952503,"teacher_disagreement_score":0.06348765,"about_ca_system_score_codex":0.0019346736,"about_ca_system_score_gemma":0.000945279,"threshold_uncertainty_score":0.09055734},"labels":[],"label_agreement":null},{"id":"W7118628814","doi":"10.1214/25-sts1001","title":"Sample-Based Planning and Learning with Function Approximation","year":2025,"lang":"","type":"article","venue":"Statistical Science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Function approximation; Core (optical fiber); Focus (optics); Dimension (graph theory); Function (biology); Approximation algorithm; Optimism","score_opus":0.015498831904994551,"score_gpt":0.28179290718513256,"score_spread":0.266294075280138,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7118628814","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005196238,0.0051174397,0.9890637,0.0003635885,0.00014882731,0.00003528734,0.000098487115,0.00029478178,0.0043583275],"genre_scores_gemma":[0.070743054,0.022195054,0.8921829,0.0007754823,0.0013552798,0.00069336867,0.00079132913,0.0006291923,0.010634386],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9984748,0.0006130185,0.00010870833,0.00029601122,0.00041910718,0.00008837882],"domain_scores_gemma":[0.9982249,0.0014070064,0.00007895041,0.0001364755,0.000104868865,0.000047905578],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024085634,0.0020518876,0.0012576412,0.0010586446,0.00038621452,0.0022541583,0.0017611169,0.0019272474,0.010957417],"category_scores_gemma":[0.00658453,0.0010462651,0.0015191025,0.0017356426,0.0022333134,0.0038222896,0.0019881546,0.003588252,0.0028716254],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007214645,0.00006205123,0.00027150594,0.00092720124,0.00008267877,0.000117884454,0.00016513887,0.08001696,0.00197843,0.72155136,0.011426811,0.18332785],"study_design_scores_gemma":[0.000028709468,0.00010425509,0.00025902942,0.00023452473,0.000032535594,0.0001539097,0.000027486907,0.18188122,0.0017139218,0.7678395,0.047678303,0.00004663612],"about_ca_topic_score_codex":0.0017270179,"about_ca_topic_score_gemma":0.0014846693,"teacher_disagreement_score":0.010957417,"about_ca_system_score_codex":0.0015111695,"about_ca_system_score_gemma":0.00096422515,"threshold_uncertainty_score":0.0366562},"labels":[],"label_agreement":null},{"id":"W7123349578","doi":"10.1109/cdc57313.2025.11312137","title":"Kernel Mean Embedding Topology: Weak and Strong Forms for Stochastic Kernels and Implications for Model Learning","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Embedding; Kernel (algebra); Robustness (evolution); Stochastic process; Network topology; Topology (electrical circuits); Weak topology (polar topology); Reproducing kernel Hilbert space","score_opus":0.03230569072649504,"score_gpt":0.3301698920966929,"score_spread":0.29786420137019787,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7123349578","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038353693,0.0005944073,0.95465857,0.0012274066,0.00006562804,0.000030377587,0.00015050874,0.00008957159,0.00482974],"genre_scores_gemma":[0.794265,0.0018320082,0.1957886,0.00049380865,0.00035235795,0.00030391465,0.0005219112,0.00020184163,0.0062405677],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99803525,0.00085549883,0.00013646048,0.00040012994,0.00042988986,0.00014281801],"domain_scores_gemma":[0.9915215,0.004537174,0.0012432497,0.00089050655,0.0010830954,0.0007246135],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004402328,0.0010401516,0.00091285724,0.0015788444,0.00076285447,0.003122069,0.0014230533,0.0017417015,0.002483649],"category_scores_gemma":[0.01604525,0.00046939604,0.0013842942,0.0011019349,0.003542061,0.008858259,0.0033489382,0.0031950106,0.0004922689],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000022670407,0.00002084041,0.0005573751,0.000043902488,0.000020494195,0.00005466917,0.00013037598,0.02099811,0.00088317686,0.9691938,0.00042283672,0.007651862],"study_design_scores_gemma":[0.000007715872,0.000047435686,0.00032437406,0.000016029368,0.0000090284275,0.00008663121,0.000060905604,0.15183954,0.00055306853,0.8449999,0.0020363687,0.000019026724],"about_ca_topic_score_codex":0.0006736886,"about_ca_topic_score_gemma":0.00050791004,"teacher_disagreement_score":0.004402328,"about_ca_system_score_codex":0.0013647337,"about_ca_system_score_gemma":0.0008086498,"threshold_uncertainty_score":0.023281991},"labels":[],"label_agreement":null},{"id":"W7124135461","doi":"10.65109/gnlj3027","title":"POMDP planning and execution in an augmented space","year":2014,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Upper and lower bounds; Markov decision process; Partially observable Markov decision process; Suite; Linear programming; Action (physics); Space (punctuation); Branch and bound","score_opus":0.022786742106857858,"score_gpt":0.27798784173079166,"score_spread":0.2552010996239338,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124135461","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04608389,0.00017451789,0.94557923,0.00025261028,0.000043414228,0.00011738196,0.00033241004,0.0021734177,0.005243092],"genre_scores_gemma":[0.55915755,0.00023715232,0.43727866,0.0000682744,0.000019048883,0.00027783617,0.00038588856,0.00015579579,0.0024196757],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999127,0.00025942404,0.0000650772,0.00017083893,0.0002425343,0.0001352845],"domain_scores_gemma":[0.99887854,0.0007072649,0.000095692885,0.0001575413,0.00011119037,0.000049719867],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009675822,0.0009522439,0.00070662843,0.00043645606,0.00058878417,0.0012622386,0.00087116647,0.0008058245,0.0033652163],"category_scores_gemma":[0.0027667736,0.00063117035,0.0009759287,0.00055130903,0.0014395282,0.0017166873,0.001493554,0.0015201841,0.00039403376],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011328278,0.00003064549,0.0003253433,0.0000880671,0.000019743524,0.00009236873,0.00009300516,0.95206016,0.001992458,0.027434133,0.0004370255,0.017313806],"study_design_scores_gemma":[0.00001976132,0.000030561256,0.0000861883,0.000012381141,0.000007901969,0.0000138083415,0.000024257793,0.9753395,0.0019801324,0.020941937,0.0015369167,0.000006645437],"about_ca_topic_score_codex":0.00886674,"about_ca_topic_score_gemma":0.009443981,"teacher_disagreement_score":0.00886674,"about_ca_system_score_codex":0.0010645868,"about_ca_system_score_gemma":0.0021087173,"threshold_uncertainty_score":0.01763022},"labels":[],"label_agreement":null},{"id":"W7124138561","doi":"10.65109/kjtg4247","title":"Efficient planning in R-max","year":2011,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Value (mathematics); Automated planning and scheduling; Markov decision process; Order (exchange); Active learning (machine learning)","score_opus":0.06178843939825616,"score_gpt":0.26707263148114885,"score_spread":0.2052841920828927,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124138561","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013625142,0.0003241243,0.9760439,0.0002863837,0.00002879841,0.00009590146,0.0001340079,0.0010437159,0.008418133],"genre_scores_gemma":[0.41047615,0.00041055537,0.58183104,0.00022510605,0.000036713012,0.00037764592,0.00038401774,0.000401791,0.0058570853],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99846137,0.000683764,0.000075323536,0.00038061079,0.00022444627,0.0001744141],"domain_scores_gemma":[0.9974075,0.0018077298,0.00019947224,0.0003221433,0.00017113259,0.00009206949],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022993411,0.0009746728,0.0012369379,0.0005290918,0.0006676781,0.001241416,0.0015440256,0.0011279313,0.006237436],"category_scores_gemma":[0.0063150227,0.0006730783,0.00093820225,0.0007825274,0.0018537403,0.0022343965,0.001989224,0.0015975191,0.0012026901],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020161585,0.00007208779,0.00040809362,0.00021606454,0.000037782807,0.00011273346,0.0001294178,0.8304871,0.0013953695,0.094679885,0.0031463823,0.06911347],"study_design_scores_gemma":[0.00003321796,0.000048946404,0.00008063259,0.000018486815,0.000010702287,0.00003155402,0.00002671654,0.90756464,0.0016071008,0.088226,0.0023410993,0.000010940423],"about_ca_topic_score_codex":0.0030742385,"about_ca_topic_score_gemma":0.0046047284,"teacher_disagreement_score":0.006237436,"about_ca_system_score_codex":0.0012836282,"about_ca_system_score_gemma":0.0025278183,"threshold_uncertainty_score":0.020866334},"labels":[],"label_agreement":null},{"id":"W7124143215","doi":"10.65109/vuye5463","title":"Optimal policy switching algorithms for reinforcement learning","year":2010,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Task (project management); Q-learning; Function (biology); Markov decision process; Function approximation; Control (management); Optimal control","score_opus":0.023923541871614384,"score_gpt":0.30137341905869686,"score_spread":0.27744987718708247,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124143215","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0053274333,0.0003838105,0.9912642,0.00018024555,0.000047102665,0.00006092341,0.000028694543,0.00032511007,0.0023824633],"genre_scores_gemma":[0.5952932,0.00078718085,0.39714304,0.00035028486,0.00012534561,0.0008193794,0.00022398547,0.00020897979,0.0050486545],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9990701,0.00040510827,0.000053302123,0.0001555213,0.00021626946,0.000099671364],"domain_scores_gemma":[0.99722785,0.0021638202,0.00017200765,0.00011070595,0.00022500836,0.00010057393],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020910555,0.0012110042,0.0014869474,0.0007998339,0.00044573183,0.0010096394,0.001578365,0.0014791039,0.004810024],"category_scores_gemma":[0.0071278308,0.0005750484,0.0005777882,0.0007205417,0.0014271077,0.0013280085,0.0014184754,0.0025238,0.0006818183],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000081447666,0.0000903973,0.0003135322,0.00008484343,0.000039116225,0.000029205426,0.00006496869,0.8759821,0.0004127384,0.059663948,0.0016259408,0.061611738],"study_design_scores_gemma":[0.000019801939,0.000016447162,0.000021024787,0.0000067243236,0.0000036339618,0.000004232908,0.000003476613,0.9759285,0.00011072878,0.023515375,0.00036636528,0.00000361763],"about_ca_topic_score_codex":0.0032600437,"about_ca_topic_score_gemma":0.0024298944,"teacher_disagreement_score":0.004810024,"about_ca_system_score_codex":0.0014829173,"about_ca_system_score_gemma":0.0015025868,"threshold_uncertainty_score":0.016091168},"labels":[],"label_agreement":null},{"id":"W7124145910","doi":"10.65109/dwjt8467","title":"Basis function discovery using spectral clustering and bisimulation metrics","year":2011,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Set (abstract data type); Feature (linguistics); Function (biology); Basis (linear algebra); Cluster analysis; Focus (optics); State (computer science)","score_opus":0.08381540224386577,"score_gpt":0.2621151350149372,"score_spread":0.17829973277107145,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124145910","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019355668,0.0004397867,0.9775039,0.00028517566,0.000025541885,0.00009477047,0.00009369656,0.00033301482,0.0018685437],"genre_scores_gemma":[0.48098975,0.000727774,0.5141692,0.00016551459,0.00006621622,0.00049923407,0.00094299763,0.00032582058,0.0021134496],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99800354,0.000918371,0.0001178413,0.00031090868,0.00051507074,0.00013429963],"domain_scores_gemma":[0.99376655,0.0035994097,0.0006016285,0.0006347086,0.001118449,0.0002792863],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003426372,0.0013461866,0.0020534436,0.0045787105,0.0012417785,0.001984705,0.0019103914,0.0020074225,0.0025327997],"category_scores_gemma":[0.019282835,0.00080928457,0.0014449505,0.0026085528,0.0013206713,0.0029611334,0.002846313,0.0017972797,0.00084753067],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013773804,0.00020607439,0.0022830563,0.0002152664,0.00013664675,0.00008166463,0.00019088906,0.69595075,0.0014854593,0.13345738,0.004065488,0.16178961],"study_design_scores_gemma":[0.0000074201193,0.000012911024,0.00009045075,0.000012821525,0.000004575354,0.000012684096,0.000013969628,0.9619054,0.00025311814,0.037242115,0.00043668397,0.000007781427],"about_ca_topic_score_codex":0.005571655,"about_ca_topic_score_gemma":0.00371123,"teacher_disagreement_score":0.005571655,"about_ca_system_score_codex":0.002074475,"about_ca_system_score_gemma":0.0022406015,"threshold_uncertainty_score":0.018120587},"labels":[],"label_agreement":null},{"id":"W7124153011","doi":"10.65109/actd7997","title":"Escaping local optima in POMDP planning as inference","year":2011,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Partially observable Markov decision process; Inference; Reinforcement learning; Controller (irrigation); Greedy algorithm; Local planning; Control (management)","score_opus":0.08219326181887587,"score_gpt":0.30553004872109796,"score_spread":0.22333678690222208,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124153011","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022409666,0.00027141135,0.97459716,0.00028462007,0.000020313031,0.000045385077,0.000023222337,0.00044851686,0.0018997359],"genre_scores_gemma":[0.7535508,0.00023992686,0.24403796,0.00020129286,0.000031273135,0.00021014932,0.000054958655,0.00012535193,0.0015482173],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99890494,0.00047305063,0.00005612054,0.00017957356,0.00025180474,0.00013453107],"domain_scores_gemma":[0.9954372,0.003593947,0.0003009536,0.0003108865,0.00021248934,0.00014450024],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031859677,0.0009618637,0.001756514,0.0006389274,0.0006255592,0.00095172005,0.0016176684,0.0013330221,0.0013700951],"category_scores_gemma":[0.009139358,0.00084913196,0.00078917603,0.0006360751,0.0026373635,0.0017804581,0.0023160174,0.0025627497,0.00022644884],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000048527214,0.000019680052,0.0002655095,0.000040671093,0.000025240448,0.000038330385,0.00007317133,0.9685348,0.0004655051,0.017664898,0.00021146834,0.012612152],"study_design_scores_gemma":[0.0000135429,0.000015785236,0.000028035647,0.00000635648,0.000005923586,0.000005003868,0.0000073012916,0.98507696,0.00026096,0.014458344,0.00011760347,0.0000042169354],"about_ca_topic_score_codex":0.006000745,"about_ca_topic_score_gemma":0.0057343943,"teacher_disagreement_score":0.006000745,"about_ca_system_score_codex":0.0013899958,"about_ca_system_score_gemma":0.0015536302,"threshold_uncertainty_score":0.01684922},"labels":[],"label_agreement":null},{"id":"W7124156211","doi":"10.65109/qppa8970","title":"Using spatial hints to improve policy reuse in a reinforcement learning agent","year":2010,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Reinforcement learning; Reuse; Exploit; Robustness (evolution); Task (project management); Domain (mathematical analysis); Policy learning","score_opus":0.031227505286058152,"score_gpt":0.3106378969221901,"score_spread":0.27941039163613196,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124156211","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.270315,0.00044417518,0.723934,0.00077922095,0.000039698578,0.00016097531,0.00004326279,0.0015983977,0.0026853397],"genre_scores_gemma":[0.909992,0.00009391601,0.08867833,0.00014813086,0.000018230374,0.00007856471,0.000033634875,0.000051440187,0.0009057808],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99830085,0.000834189,0.00011121681,0.00028141064,0.00032088763,0.0001514451],"domain_scores_gemma":[0.9883746,0.007915472,0.0012061707,0.0011974082,0.00078954885,0.00051686954],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00366315,0.0013806676,0.0013162675,0.00064634555,0.00048810832,0.0008032994,0.0016795612,0.001683852,0.001397253],"category_scores_gemma":[0.019335737,0.00060300983,0.0004498346,0.00042138464,0.0014742806,0.0019587462,0.001604306,0.0014279562,0.00033092822],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00085669616,0.00071065663,0.0056179115,0.00023219523,0.00019671004,0.00038109007,0.0008192048,0.81875855,0.013483182,0.00871531,0.0008595171,0.14936897],"study_design_scores_gemma":[0.00011792342,0.00029323203,0.0004196751,0.000020402007,0.00004558583,0.00006730393,0.00005016144,0.9851698,0.005577357,0.0075565698,0.00065164594,0.000030374242],"about_ca_topic_score_codex":0.0031385648,"about_ca_topic_score_gemma":0.0033984561,"teacher_disagreement_score":0.00366315,"about_ca_system_score_codex":0.00077087106,"about_ca_system_score_gemma":0.0015474468,"threshold_uncertainty_score":0.01937282},"labels":[],"label_agreement":null},{"id":"W7124164391","doi":"10.65109/pwam3332","title":"Smart exploration in reinforcement learning using absolute temporal difference errors","year":2013,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; State (computer science); Temporal difference learning; Function (biology); Control (management); Function approximation","score_opus":0.060067349247302546,"score_gpt":0.2738622850019143,"score_spread":0.21379493575461173,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124164391","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03218501,0.00023431366,0.9663579,0.00011749255,0.00002877618,0.000024021512,0.000012382572,0.00018892776,0.0008513137],"genre_scores_gemma":[0.8937624,0.00016017741,0.104462504,0.00006288031,0.00003526307,0.00010738646,0.000035681973,0.000071589064,0.001302137],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990011,0.00041981545,0.00006219536,0.00016349423,0.00027453183,0.00007900253],"domain_scores_gemma":[0.9944174,0.0042573293,0.0004510807,0.00030984677,0.00034860836,0.00021576707],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026718203,0.0008363477,0.0010906828,0.0004913798,0.00025981313,0.00086528325,0.0011360056,0.00087247853,0.0011266983],"category_scores_gemma":[0.010996159,0.00044014724,0.00040234654,0.00041288527,0.0016878126,0.002105606,0.0015794345,0.0014686288,0.00014989785],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018600418,0.00005027924,0.00085604447,0.00007903602,0.000036424506,0.00005478567,0.00007836601,0.92107314,0.0020874399,0.0361036,0.00026356798,0.039131362],"study_design_scores_gemma":[0.000012405273,0.000027287137,0.000047241967,0.0000035800479,0.00000242637,0.0000062455247,0.0000018465117,0.9915758,0.0003911631,0.007850587,0.000078096426,0.0000033091849],"about_ca_topic_score_codex":0.0018888095,"about_ca_topic_score_gemma":0.001291005,"teacher_disagreement_score":0.0026718203,"about_ca_system_score_codex":0.00083985797,"about_ca_system_score_gemma":0.0008503809,"threshold_uncertainty_score":0.0141301155},"labels":[],"label_agreement":null},{"id":"W7124164408","doi":"10.65109/ycuv1950","title":"Policy optimization by marginal-map probabilistic inference in generative models","year":2014,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Partially observable Markov decision process; Inference; Benchmark (surveying); Bounded function; Generative model; Probabilistic logic; Bayesian inference; Scalability","score_opus":0.022099080594102625,"score_gpt":0.26834929519673417,"score_spread":0.24625021460263155,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124164408","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005259996,0.0000919797,0.99360883,0.00010837185,0.000009728523,0.000016276092,0.00003068004,0.00019764714,0.0006764846],"genre_scores_gemma":[0.68675524,0.00029128962,0.3101177,0.00018055252,0.000053989632,0.00019541307,0.00021467025,0.00022115554,0.0019700367],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99909556,0.00034175592,0.000036560465,0.00019372496,0.00022611038,0.000106379324],"domain_scores_gemma":[0.99723345,0.0021681094,0.00017419529,0.00017660786,0.00015592235,0.00009168685],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021087553,0.0009894794,0.0015129059,0.0006737181,0.0005223274,0.0013277619,0.0020434337,0.0013728243,0.0022990096],"category_scores_gemma":[0.007713744,0.0009407798,0.0011993931,0.0007723023,0.002141656,0.0015954222,0.0022031476,0.0023886543,0.0002916769],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000022194934,0.000012489879,0.00018259902,0.000029838619,0.00001578006,0.000019414101,0.000031308264,0.974934,0.00025700734,0.017149175,0.00021680344,0.007129354],"study_design_scores_gemma":[0.0000034137715,0.0000039789006,0.000019722183,0.0000022997363,0.0000022977588,0.0000035306516,0.0000030466497,0.9912339,0.00011900647,0.008514163,0.000092446986,0.0000022314775],"about_ca_topic_score_codex":0.0114654135,"about_ca_topic_score_gemma":0.010141458,"teacher_disagreement_score":0.0114654135,"about_ca_system_score_codex":0.0018949095,"about_ca_system_score_gemma":0.0023410432,"threshold_uncertainty_score":0.022797346},"labels":[],"label_agreement":null},{"id":"W7124165107","doi":"10.65109/twxw5772","title":"Sigma point policy iteration","year":2008,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Iterated function; Sigma; Function (biology); Fixed point; Point (geometry); Order (exchange); Markov decision process; Bellman equation","score_opus":0.028661473987990967,"score_gpt":0.2603251415755973,"score_spread":0.23166366758760631,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124165107","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005887225,0.00013148744,0.9912964,0.00007601561,0.000039621627,0.000051963467,0.000016286414,0.00032612585,0.0021748368],"genre_scores_gemma":[0.44599116,0.00029158866,0.54394,0.000284304,0.000055905846,0.00061403896,0.00018844468,0.00024849208,0.008386105],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9994704,0.00017297681,0.000035014844,0.000094220275,0.00016519529,0.00006218648],"domain_scores_gemma":[0.998442,0.001012106,0.00009982313,0.00011106717,0.00027694507,0.000058072208],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011857802,0.0008170186,0.0013338532,0.00046942153,0.00042567414,0.0009118154,0.0009307664,0.0012390533,0.004990209],"category_scores_gemma":[0.0043850876,0.0005301931,0.00061725534,0.00048462683,0.0010945392,0.00084337173,0.0014231956,0.0013481196,0.0011213596],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000159808,0.00008855011,0.000907106,0.00016793508,0.00006067259,0.00007648698,0.00016565472,0.7700327,0.0026229178,0.051113855,0.002174982,0.17242922],"study_design_scores_gemma":[0.00001359242,0.000033653338,0.000031725143,0.000011132994,0.0000040414698,0.000013911001,0.000008219773,0.989594,0.0007445714,0.008758645,0.0007814543,0.000005001016],"about_ca_topic_score_codex":0.003052514,"about_ca_topic_score_gemma":0.0023029593,"teacher_disagreement_score":0.004990209,"about_ca_system_score_codex":0.0008022429,"about_ca_system_score_gemma":0.0014962379,"threshold_uncertainty_score":0.01669395},"labels":[],"label_agreement":null},{"id":"W7124167367","doi":"10.65109/hsjq2262","title":"Distributed multiagent resource allocation with adaptive preemption for dynamic tasks","year":2014,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Preemption; Resource allocation; Task (project management); Resource (disambiguation); Multi-agent system","score_opus":0.01697424960136613,"score_gpt":0.24673887685720808,"score_spread":0.22976462725584196,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124167367","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011744004,0.00018772698,0.9864153,0.00008167069,0.000041454896,0.000033948272,0.000006612263,0.00019725508,0.001292072],"genre_scores_gemma":[0.7723572,0.00018111616,0.2237289,0.00011760709,0.00006838621,0.00017598752,0.000026848753,0.00006893887,0.0032748878],"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","domain_scores_codex":[0.999313,0.00024359903,0.00004077328,0.00015435618,0.0001612859,0.00008695695],"domain_scores_gemma":[0.9990754,0.0004282975,0.00011870514,0.00017282684,0.000113399576,0.00009129083],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010974386,0.0005881428,0.0006453213,0.00026569422,0.0005007619,0.00066765194,0.0014976298,0.0005703665,0.001388095],"category_scores_gemma":[0.0025568667,0.00029063012,0.00039920377,0.00029058856,0.0006464694,0.0010198662,0.0012742548,0.0011434305,0.0002757443],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023410322,0.00019000011,0.0005569304,0.00014374057,0.00008190341,0.00022754767,0.00020203469,0.8496069,0.014969143,0.031997614,0.0013154004,0.1004747],"study_design_scores_gemma":[0.00002934703,0.000049930404,0.00009752582,0.0000048604584,0.000010607362,0.000038754966,0.00001390657,0.9862445,0.0015320382,0.010345554,0.0016265282,0.0000063927614],"about_ca_topic_score_codex":0.0010683606,"about_ca_topic_score_gemma":0.0012815857,"teacher_disagreement_score":0.0014976298,"about_ca_system_score_codex":0.0004965672,"about_ca_system_score_gemma":0.0007777913,"threshold_uncertainty_score":0.005803883},"labels":[],"label_agreement":null},{"id":"W7124174011","doi":"10.65109/zzer3937","title":"Using bisimulation for policy transfer in MDPs","year":2010,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Markov decision process; Markov process; Work (physics); Bisimulation; Transfer (computing); Partially observable Markov decision process","score_opus":0.06793892293769213,"score_gpt":0.355023856049797,"score_spread":0.2870849331121049,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124174011","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006755833,0.00022463127,0.98896563,0.00024817674,0.000042838234,0.000072734496,0.000055415967,0.0002922745,0.003342406],"genre_scores_gemma":[0.64510065,0.0008741588,0.34589708,0.00045395936,0.00011577266,0.0013124273,0.00035224063,0.0004926299,0.0054012057],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975744,0.0013219896,0.00015141805,0.0003883643,0.00037681693,0.00018695778],"domain_scores_gemma":[0.98952883,0.008418654,0.00064788974,0.0005800743,0.00052268145,0.00030185928],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004874342,0.002062394,0.0022873026,0.0013482762,0.0008813567,0.0017505126,0.002140412,0.0023234983,0.00769304],"category_scores_gemma":[0.019381646,0.0011707079,0.0015971267,0.0010685917,0.0027146705,0.0035309421,0.0043460145,0.0036997825,0.0011134445],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004722316,0.00003546599,0.00016048273,0.000076807046,0.000034055534,0.000038511702,0.00005932235,0.91157603,0.00029112175,0.07577798,0.00032436955,0.011578611],"study_design_scores_gemma":[0.000016468877,0.000023168073,0.000013589499,0.000014296949,0.000005760701,0.0000064128,0.000005943248,0.94140756,0.00018232645,0.057820823,0.00049679604,0.0000068261297],"about_ca_topic_score_codex":0.0043155383,"about_ca_topic_score_gemma":0.0032888127,"teacher_disagreement_score":0.00769304,"about_ca_system_score_codex":0.0025470608,"about_ca_system_score_gemma":0.0025943988,"threshold_uncertainty_score":0.025778353},"labels":[],"label_agreement":null},{"id":"W7124176698","doi":"10.65109/anfh7318","title":"Incremental Policy Iteration with Guaranteed Escape from Local Optima in POMDP Planning","year":2015,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Partially observable Markov decision process; Bounded function; Controller (irrigation); Markov decision process; Scale (ratio); Energy consumption; Property (philosophy); Local search (optimization)","score_opus":0.0358786020138582,"score_gpt":0.28255704684655975,"score_spread":0.24667844483270154,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124176698","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021270456,0.00031441418,0.9727011,0.0002795967,0.000036188867,0.000083971536,0.000046136734,0.00073256536,0.004535546],"genre_scores_gemma":[0.79326576,0.00026768784,0.20247844,0.00023934818,0.00004094448,0.0005012015,0.0001346331,0.0002083147,0.002863563],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99865216,0.0005818142,0.00006692737,0.00023180744,0.00029897125,0.00016846797],"domain_scores_gemma":[0.9939627,0.0050168205,0.0003323937,0.00022646041,0.00025639875,0.00020520527],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024402323,0.0013534378,0.0019217373,0.00075222045,0.00063307834,0.0010797359,0.00160867,0.0014792024,0.00266202],"category_scores_gemma":[0.008205039,0.0008849027,0.000975041,0.0005799413,0.002416921,0.0014588243,0.0023347894,0.0025001026,0.00037573883],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007245771,0.000041437695,0.00025710496,0.00006832103,0.00002922234,0.000062141546,0.000079746824,0.9715612,0.00032673034,0.014798319,0.000485513,0.012217832],"study_design_scores_gemma":[0.000017307262,0.00002432995,0.00002540165,0.000007295144,0.0000049123364,0.0000077226205,0.0000063243465,0.9906596,0.00016290651,0.008889052,0.00019065126,0.0000044953727],"about_ca_topic_score_codex":0.005880076,"about_ca_topic_score_gemma":0.005630396,"teacher_disagreement_score":0.005880076,"about_ca_system_score_codex":0.0013773966,"about_ca_system_score_gemma":0.0025255806,"threshold_uncertainty_score":0.0129053},"labels":[],"label_agreement":null},{"id":"W7124178739","doi":"10.65109/muso7637","title":"Quasi deterministic POMDPs and DecPOMDPs","year":2010,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Observability; Observable; Extension (predicate logic); Class (philosophy); Markov decision process; Markov chain; Markov process","score_opus":0.012205200261922752,"score_gpt":0.25202832314467016,"score_spread":0.2398231228827474,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124178739","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034930438,0.00053405855,0.9544658,0.00064076215,0.00007621749,0.000073175834,0.0007416109,0.0002564436,0.0082814805],"genre_scores_gemma":[0.8323301,0.0009382372,0.15482448,0.0002939185,0.00009788331,0.0002850707,0.0009298245,0.00009230026,0.010208122],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99911755,0.00022725723,0.000055525208,0.0002474408,0.0002200421,0.00013221822],"domain_scores_gemma":[0.9973717,0.0015903547,0.00042731888,0.00026780897,0.0002046569,0.00013819162],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010021457,0.0006504073,0.00072355283,0.00034904887,0.0004727215,0.0011830257,0.00082016963,0.00081542274,0.00450488],"category_scores_gemma":[0.0035825043,0.0004882295,0.0007347174,0.00052501477,0.0011261473,0.002054113,0.0012173303,0.0017347183,0.00028810994],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000086575565,0.000055247525,0.0012356769,0.00016885134,0.000050541956,0.00021722769,0.00009528798,0.50082976,0.001208271,0.47843474,0.0016145571,0.01600322],"study_design_scores_gemma":[0.000019510306,0.000033731045,0.00032661416,0.000013697347,0.000010718614,0.00006177674,0.000032050204,0.78264016,0.0004559958,0.21286662,0.0035294923,0.000009709103],"about_ca_topic_score_codex":0.003625266,"about_ca_topic_score_gemma":0.0041488297,"teacher_disagreement_score":0.00450488,"about_ca_system_score_codex":0.0012253918,"about_ca_system_score_gemma":0.0012980315,"threshold_uncertainty_score":0.015070379},"labels":[],"label_agreement":null},{"id":"W7124181186","doi":"10.65109/apft3995","title":"Expectation-Maximization for Inverse Reinforcement Learning with Hidden Data","year":2016,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Missing data; Reinforcement learning; Task (project management); Trajectory; Feature (linguistics); Key (lock); State space; State (computer science); Sorting","score_opus":0.049844816202744585,"score_gpt":0.27516061794495544,"score_spread":0.22531580174221086,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124181186","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0040756217,0.00017912805,0.99463516,0.00018646134,0.000022724716,0.000023967546,0.000017776872,0.00017335943,0.00068568904],"genre_scores_gemma":[0.61023736,0.00040263348,0.3847301,0.00035454056,0.00010291643,0.000407977,0.00018993439,0.00019356086,0.0033809387],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987482,0.00062997785,0.000066386194,0.00022393017,0.00021865733,0.000112756905],"domain_scores_gemma":[0.99530584,0.003731777,0.00026596722,0.00019723635,0.000352435,0.00014669624],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033640729,0.0014144953,0.0019326968,0.0004436492,0.00035359646,0.0010422419,0.0019405974,0.00159506,0.0021789777],"category_scores_gemma":[0.010852676,0.0006163744,0.0007789339,0.0006387633,0.0018563176,0.0013458418,0.0016428055,0.0026570002,0.00055723893],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008628339,0.00006658535,0.00036375786,0.00009954531,0.000056988298,0.00007875861,0.00006576426,0.9523689,0.0005753265,0.019513981,0.00081048015,0.025913682],"study_design_scores_gemma":[0.000009709158,0.000016668433,0.000021342299,0.000004830267,0.000003504512,0.0000058757073,0.0000029375271,0.9898017,0.00018877843,0.009764059,0.00017728613,0.0000033610027],"about_ca_topic_score_codex":0.0034276552,"about_ca_topic_score_gemma":0.0027707126,"teacher_disagreement_score":0.0034276552,"about_ca_system_score_codex":0.0013557696,"about_ca_system_score_gemma":0.0016440605,"threshold_uncertainty_score":0.017791092},"labels":[],"label_agreement":null},{"id":"W7124207829","doi":"10.65109/ccmg9392","title":"From Global Selective Perception to Local Selective Perception","year":2004,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Reinforcement learning; Perception; Active perception; Reinforcement; Work (physics); Control (management)","score_opus":0.012729836536510363,"score_gpt":0.26972374028830953,"score_spread":0.25699390375179915,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124207829","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017555585,0.0010324604,0.96998143,0.00072165043,0.000068769245,0.000025581876,0.00002167181,0.00033492147,0.010257886],"genre_scores_gemma":[0.81403106,0.0012297521,0.17803584,0.0005129064,0.00013362033,0.000090051144,0.000059588776,0.00016272135,0.0057444954],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99950147,0.00014330418,0.000016777465,0.00014381211,0.00012308975,0.00007152589],"domain_scores_gemma":[0.99899834,0.0004987425,0.00008860209,0.00018598001,0.0001222217,0.00010619453],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008149535,0.0005600599,0.00045210033,0.00024513612,0.0003844425,0.0011931236,0.0009775218,0.000656574,0.0020361885],"category_scores_gemma":[0.0020778042,0.0003064373,0.00042443204,0.00024824642,0.002517799,0.0026250314,0.0019027167,0.0016688624,0.00032381876],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003349531,0.000102059785,0.0014889822,0.00032666014,0.00011898159,0.00022209722,0.0007863502,0.2144727,0.017178342,0.50691515,0.004155467,0.2538983],"study_design_scores_gemma":[0.000052793886,0.0001730578,0.0006605568,0.0000392921,0.000053211603,0.00012132019,0.00012628714,0.6241517,0.0064622215,0.35881054,0.009314811,0.000034299854],"about_ca_topic_score_codex":0.0014306185,"about_ca_topic_score_gemma":0.0011114163,"teacher_disagreement_score":0.0020361885,"about_ca_system_score_codex":0.0006655209,"about_ca_system_score_gemma":0.0005556997,"threshold_uncertainty_score":0.006811738},"labels":[],"label_agreement":null},{"id":"W7124240193","doi":"10.65109/jsvi2318","title":"FedFormer: Contextual Federation with Attention in Reinforcement Learning","year":2023,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Aggregate (composite); Reinforcement; Transformer","score_opus":0.02335311171161477,"score_gpt":0.2575649258979304,"score_spread":0.23421181418631562,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124240193","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015008921,0.00020371756,0.9813426,0.00021745618,0.000059717357,0.00006345929,0.000035332203,0.0016969609,0.0013719231],"genre_scores_gemma":[0.8012729,0.00011317811,0.19529082,0.00040501912,0.00006788482,0.00020413619,0.00014515362,0.0002072617,0.0022935735],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987632,0.00050419854,0.000056938417,0.00032032916,0.00020571925,0.0001495953],"domain_scores_gemma":[0.9978154,0.0010530856,0.00019477002,0.0004715854,0.00028633163,0.0001788869],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033129028,0.001250659,0.00147439,0.0005288462,0.00066972635,0.001244487,0.002470305,0.001498722,0.002693845],"category_scores_gemma":[0.0072646793,0.0006156171,0.0006451981,0.0005327853,0.0014508934,0.0024114386,0.0029955192,0.0020768347,0.00053096586],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023829578,0.00021180825,0.0019343088,0.000112617716,0.00013847237,0.00015607949,0.00023127122,0.8239722,0.0033467487,0.015786022,0.0032309706,0.1506412],"study_design_scores_gemma":[0.000016543985,0.000037503756,0.00007263702,0.000007026307,0.000010300962,0.000016404709,0.000011680959,0.99189746,0.0007317172,0.0067164367,0.00047693754,0.000005296941],"about_ca_topic_score_codex":0.0045187864,"about_ca_topic_score_gemma":0.0051811086,"teacher_disagreement_score":0.0045187864,"about_ca_system_score_codex":0.0010818915,"about_ca_system_score_gemma":0.0017782392,"threshold_uncertainty_score":0.017520487},"labels":[],"label_agreement":null},{"id":"W7124241411","doi":"10.65109/ldqz4728","title":"Learning from Multiple Independent Advisors in Multi-agent Reinforcement Learning","year":2023,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; University of Alberta; Vector Institute","funders":"","keywords":"Reinforcement learning; Sample complexity; Set (abstract data type); Sample (material); Convergence (economics); Action (physics); Error-driven learning; State (computer science)","score_opus":0.05499245184937777,"score_gpt":0.27924540457871294,"score_spread":0.22425295272933518,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124241411","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03463811,0.00030498946,0.9628806,0.00033886085,0.000032028147,0.00012206485,0.00003101188,0.0005563411,0.0010960497],"genre_scores_gemma":[0.8469832,0.00012443138,0.15054505,0.0002731439,0.000044961827,0.00023300067,0.00007382651,0.00007138947,0.0016510022],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99818903,0.00079842424,0.00010280633,0.00040083463,0.00030493527,0.00020394963],"domain_scores_gemma":[0.99291563,0.0048919036,0.00070625165,0.0004056072,0.00063300773,0.00044753245],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046497067,0.0014315499,0.0022314475,0.0006387405,0.00066428335,0.0010451324,0.0023174281,0.0021538853,0.0016712428],"category_scores_gemma":[0.011689594,0.0008519909,0.0005538057,0.0005929685,0.0018989841,0.0018670949,0.0016921117,0.0025844406,0.00036911815],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001510548,0.00011888459,0.0016258302,0.00006997721,0.000056810393,0.00008693039,0.00009373591,0.94429225,0.0006229182,0.0069509237,0.00055017567,0.045380507],"study_design_scores_gemma":[0.0000270867,0.00003578593,0.0000813674,0.0000068751788,0.000007777845,0.000012094125,0.000006549602,0.994584,0.00029117422,0.0048002005,0.00014108964,0.000006007777],"about_ca_topic_score_codex":0.0053279786,"about_ca_topic_score_gemma":0.0058150534,"teacher_disagreement_score":0.0053279786,"about_ca_system_score_codex":0.0012873906,"about_ca_system_score_gemma":0.0019400461,"threshold_uncertainty_score":0.024590313},"labels":[],"label_agreement":null},{"id":"W7124241732","doi":"10.65109/xgyf2237","title":"Interpretable Preference-based Reinforcement Learning with Tree-Structured Reward Functions","year":2022,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Thales (Canada)","funders":"","keywords":"Interpretability; Reinforcement learning; Robustness (evolution); Function (biology); Heuristic; Preference","score_opus":0.028391592682051756,"score_gpt":0.21991690947366765,"score_spread":0.19152531679161588,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124241732","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.054373257,0.00008059584,0.94362974,0.00019994058,0.000019554798,0.000050657138,0.000050407314,0.00042547728,0.0011702785],"genre_scores_gemma":[0.8551107,0.00005587826,0.14338744,0.00011411933,0.000012807156,0.00012199312,0.000075857206,0.0000881502,0.0010330658],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991666,0.00042172562,0.000041855405,0.00014104029,0.00014244829,0.00008636318],"domain_scores_gemma":[0.9954125,0.003148538,0.00044482772,0.000364702,0.0004266727,0.00020267193],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018951253,0.00078228285,0.000878378,0.0003794435,0.00035091917,0.0008903412,0.0010932091,0.0011373471,0.0021019462],"category_scores_gemma":[0.01222292,0.00045570085,0.00046106888,0.0003452955,0.001432289,0.0015981551,0.0010791272,0.0019086784,0.00034389584],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016482023,0.00009216411,0.0013388722,0.00007185111,0.000026408557,0.00009359415,0.00013913393,0.9286723,0.003203019,0.027776843,0.0006258684,0.037795167],"study_design_scores_gemma":[0.00001422376,0.000026352911,0.00005745327,0.000003831933,0.0000024800493,0.000006908849,0.000004587471,0.98800874,0.00041852274,0.011366412,0.000086592154,0.0000039870047],"about_ca_topic_score_codex":0.0020616758,"about_ca_topic_score_gemma":0.0026468653,"teacher_disagreement_score":0.0021019462,"about_ca_system_score_codex":0.0009670716,"about_ca_system_score_gemma":0.0011360417,"threshold_uncertainty_score":0.010022521},"labels":[],"label_agreement":null},{"id":"W7124242323","doi":"10.65109/kkdn1922","title":"Dynamic Reward Sharing to Enhance Learning in the Context of Multiagent Teams","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Process (computing); Context (archaeology); Function (biology); Social learning; Hyperparameter; Policy learning","score_opus":0.01162814992993511,"score_gpt":0.30194284473227656,"score_spread":0.29031469480234146,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124242323","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08550436,0.00027412004,0.90968406,0.00036852932,0.000046846166,0.00007362343,0.000018951116,0.00029952638,0.0037299441],"genre_scores_gemma":[0.95783865,0.00006459819,0.040832285,0.00007285503,0.00002281114,0.000069571346,0.000012314704,0.000033931276,0.0010530442],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988244,0.0005719948,0.00004231321,0.00022978992,0.00019117247,0.00014033909],"domain_scores_gemma":[0.99688596,0.0016660643,0.0005201778,0.00033432004,0.00024648116,0.00034707572],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002251573,0.00075123797,0.00093339116,0.00033517464,0.0005352599,0.00084564625,0.0013392187,0.0008337271,0.0016407697],"category_scores_gemma":[0.008614484,0.00032119974,0.00038680146,0.00024670025,0.0013894804,0.0018144187,0.0025263953,0.0012727167,0.0002699904],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012447816,0.00022587275,0.0015462835,0.000069918075,0.00006348555,0.00013596605,0.00025840435,0.9041966,0.0055880425,0.037386954,0.0007234286,0.049680527],"study_design_scores_gemma":[0.00002385036,0.00008578571,0.0002005333,0.000007391281,0.000009781497,0.000020532238,0.000020061678,0.97289586,0.0010000613,0.025229633,0.00049849396,0.000008051643],"about_ca_topic_score_codex":0.0010109089,"about_ca_topic_score_gemma":0.0010596531,"teacher_disagreement_score":0.002251573,"about_ca_system_score_codex":0.0008328573,"about_ca_system_score_gemma":0.0009876982,"threshold_uncertainty_score":0.011907637},"labels":[],"label_agreement":null},{"id":"W7124247083","doi":"10.65109/egea9379","title":"β-DQN: Improving Deep Q-Learning By Evolving the Behavior","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Alberta","funders":"","keywords":"Generality; Simple (philosophy); Function (biology); Range (aeronautics); Population; Overhead (engineering)","score_opus":0.0075758198450625105,"score_gpt":0.24484235876922628,"score_spread":0.23726653892416377,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124247083","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028552568,0.00052309665,0.9670547,0.00027952404,0.00008807618,0.000087599285,0.00007112045,0.0011744621,0.0021688414],"genre_scores_gemma":[0.70197225,0.00039835396,0.2925726,0.00077506923,0.00007329643,0.00031484218,0.00029870382,0.00034263515,0.0032523144],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99929297,0.00021069357,0.000043993143,0.00018936895,0.00017159875,0.00009139469],"domain_scores_gemma":[0.9976285,0.0014343766,0.00018878185,0.00024688643,0.00034698096,0.00015452129],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022705,0.0010201236,0.0010675981,0.00052740955,0.0003378103,0.00059096044,0.0022820332,0.0010020398,0.002291459],"category_scores_gemma":[0.008837788,0.00054539595,0.00049991993,0.00041324736,0.0010625295,0.0015413155,0.0017687399,0.0017981456,0.00049510796],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017549595,0.00023243275,0.0040963055,0.00018123766,0.00007931146,0.00008819199,0.0001438511,0.74344414,0.0041941823,0.011544907,0.003568431,0.23225152],"study_design_scores_gemma":[0.000019005618,0.000044565535,0.00008650797,0.00001168574,0.0000068097343,0.00001580913,0.0000058834953,0.9941922,0.0004722506,0.004625226,0.00051618204,0.000003956054],"about_ca_topic_score_codex":0.0047017876,"about_ca_topic_score_gemma":0.0057002665,"teacher_disagreement_score":0.0047017876,"about_ca_system_score_codex":0.0009669841,"about_ca_system_score_gemma":0.0019315557,"threshold_uncertainty_score":0.012007713},"labels":[],"label_agreement":null},{"id":"W7124247225","doi":"10.65109/vzsi9543","title":"Hiking up that HILL with Cogment-Verse: Train &amp; Operate Multi-agent Systems Learning from Humans","year":2023,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Network for Business Sustainability; Institut national de psychiatrie légale Philippe-Pinel","funders":"","keywords":"Variety (cybernetics); Reinforcement learning; Generalization; Context (archaeology); Formalism (music); Applications of artificial intelligence","score_opus":0.12597040934911097,"score_gpt":0.290253081006849,"score_spread":0.16428267165773802,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124247225","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0074855536,0.00027081912,0.92124444,0.0022273923,0.00033219066,0.00018923916,0.0004312871,0.028253412,0.03956568],"genre_scores_gemma":[0.17503956,0.0004682051,0.7804693,0.0014096228,0.00010188794,0.0005023501,0.001034099,0.007107737,0.0338672],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99949145,0.00016095911,0.000026576003,0.0001220831,0.00013836613,0.00006060579],"domain_scores_gemma":[0.9990404,0.00036736063,0.000054648735,0.00032189803,0.000092337454,0.0001232715],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013561525,0.00085157086,0.00037616174,0.0004003274,0.0007122393,0.0019003819,0.0017530251,0.001635472,0.021754887],"category_scores_gemma":[0.004926422,0.00047730666,0.00059712364,0.00024423024,0.0024456508,0.0029475077,0.0038308783,0.0023855395,0.005313823],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051874,0.00032110966,0.001755926,0.0006763419,0.00010707724,0.00048335144,0.0017888857,0.17246683,0.017480798,0.34401962,0.11238444,0.34799683],"study_design_scores_gemma":[0.00012939422,0.0001878566,0.0007019712,0.000299482,0.000031820684,0.0003033124,0.00018663428,0.4225765,0.016866656,0.17956083,0.37905222,0.000103414444],"about_ca_topic_score_codex":0.0029829594,"about_ca_topic_score_gemma":0.0059008757,"teacher_disagreement_score":0.021754887,"about_ca_system_score_codex":0.00072483334,"about_ca_system_score_gemma":0.0011666046,"threshold_uncertainty_score":0.07277733},"labels":[],"label_agreement":null},{"id":"W7124253663","doi":"10.65109/rjqi2619","title":"Off-Policy Evolutionary Reinforcement Learning with Maximum Mutations","year":2022,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Scalability; Hyperparameter; Learning classifier system; Population; Evolutionary algorithm; Evolutionary computation; Sensitivity (control systems)","score_opus":0.015079677678669174,"score_gpt":0.24667902275559514,"score_spread":0.23159934507692598,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124253663","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038558345,0.00034132387,0.9528678,0.00028786543,0.0000926114,0.00009305566,0.000037840407,0.0008381098,0.006882971],"genre_scores_gemma":[0.88450265,0.00014282847,0.109174155,0.00024968974,0.00004242113,0.00023416107,0.000072425646,0.00013925903,0.0054424247],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99939287,0.00021949003,0.000026649237,0.00010211508,0.00017347664,0.00008539002],"domain_scores_gemma":[0.9979925,0.0013142591,0.00017390195,0.00018839883,0.0002095179,0.00012139978],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013310564,0.0009923007,0.0011829169,0.0005413817,0.0003691558,0.0007746307,0.0013735495,0.0010868334,0.0031718083],"category_scores_gemma":[0.0057180235,0.00045381146,0.00046023197,0.00039593698,0.0011601245,0.0007732917,0.0014105377,0.0012453425,0.00060153916],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008768242,0.000059840822,0.0005792799,0.000044600612,0.000037920945,0.00009873388,0.000051280967,0.93928784,0.0015551059,0.013511632,0.00095777825,0.04372836],"study_design_scores_gemma":[0.000015447558,0.00001960805,0.000041066673,0.0000054185034,0.0000039270585,0.000010668552,0.00000342042,0.9955075,0.00025489772,0.0038530957,0.00028165246,0.0000033649803],"about_ca_topic_score_codex":0.0020557824,"about_ca_topic_score_gemma":0.0022810597,"teacher_disagreement_score":0.0031718083,"about_ca_system_score_codex":0.0007804705,"about_ca_system_score_gemma":0.0009889548,"threshold_uncertainty_score":0.010610759},"labels":[],"label_agreement":null},{"id":"W7124257082","doi":"10.65109/ehmm3042","title":"Search-Improved Game-Theoretic Multiagent Reinforcement Learning in General and Negotiation Games","year":2023,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Negotiation; Generative grammar; Bayesian probability; Test (biology); Representation (politics); Bayesian inference; Social learning; Reinforcement","score_opus":0.0221885923684442,"score_gpt":0.27029056563293824,"score_spread":0.24810197326449404,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124257082","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1555994,0.00027405875,0.8392604,0.0004208236,0.000032999258,0.00010790249,0.00005121808,0.00032745936,0.0039258176],"genre_scores_gemma":[0.9254081,0.000065978806,0.07332395,0.00007357268,0.000012162813,0.000106577405,0.000050085913,0.000032467713,0.0009272473],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99919075,0.0004942573,0.000030856747,0.000105760635,0.000097043274,0.00008132694],"domain_scores_gemma":[0.9959735,0.0030275434,0.00032660787,0.00023962623,0.00027096402,0.00016177655],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029564956,0.00061959855,0.0011144382,0.0005061921,0.00042264524,0.0007853115,0.0014755481,0.00076840416,0.0016028717],"category_scores_gemma":[0.009186825,0.00039293358,0.00051463296,0.00041545797,0.0012312134,0.001241866,0.0012349848,0.0012930459,0.00019820027],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000033506883,0.00003694954,0.0006378268,0.000018795157,0.000019321817,0.000017959574,0.00003771652,0.98258024,0.00019842152,0.0073673446,0.00017358999,0.008878319],"study_design_scores_gemma":[0.0000079974,0.000013559788,0.00004725435,0.0000015969224,0.0000021480867,0.000003113999,0.0000028012964,0.9968701,0.000070897135,0.002914901,0.00006405505,0.0000016144865],"about_ca_topic_score_codex":0.005555228,"about_ca_topic_score_gemma":0.005259837,"teacher_disagreement_score":0.005555228,"about_ca_system_score_codex":0.0010948621,"about_ca_system_score_gemma":0.001264678,"threshold_uncertainty_score":0.01563561},"labels":[],"label_agreement":null},{"id":"W7124264120","doi":"10.65109/zliw2815","title":"Mastering Robot Control through Point-based Reinforcement Learning with Pre-training","year":2024,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Robustness (evolution); Robot; Leverage (statistics); Robotics; Limiting; Rendering (computer graphics)","score_opus":0.026406348197170807,"score_gpt":0.2616624150467427,"score_spread":0.2352560668495719,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124264120","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031961057,0.00018994058,0.9644992,0.00012986452,0.0000399132,0.00008092785,0.000016757635,0.001233236,0.001849093],"genre_scores_gemma":[0.88697857,0.00011921288,0.11101939,0.00016006775,0.000028966795,0.0001729806,0.00006456568,0.00008330071,0.0013730245],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99951196,0.00010783424,0.000025526871,0.00013390988,0.0001384537,0.00008236561],"domain_scores_gemma":[0.9977703,0.0014157261,0.00023124588,0.00021511073,0.00026071878,0.000106889434],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010371257,0.001293846,0.0009598927,0.00029112323,0.00032016332,0.0005946276,0.0017061878,0.0010324223,0.0018704481],"category_scores_gemma":[0.0048834216,0.0005990312,0.0004673296,0.0002096243,0.0010927572,0.0010336323,0.0012931562,0.002045214,0.00036753717],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010788059,0.000094013856,0.0006497996,0.000066271525,0.000024098195,0.00007022321,0.000060717506,0.93197423,0.00439193,0.0021568001,0.0005094089,0.059894647],"study_design_scores_gemma":[0.000008882959,0.00004008393,0.000054736953,0.000004249306,0.0000031795466,0.000008643381,0.0000028901027,0.99818075,0.00082663854,0.00076493353,0.00010171459,0.000003334479],"about_ca_topic_score_codex":0.0052413223,"about_ca_topic_score_gemma":0.005097632,"teacher_disagreement_score":0.0052413223,"about_ca_system_score_codex":0.0006234611,"about_ca_system_score_gemma":0.0012173401,"threshold_uncertainty_score":0.010421634},"labels":[],"label_agreement":null},{"id":"W7124267618","doi":"10.65109/odcq7385","title":"Centralized Model and Exploration Policy for Multi-Agent RL","year":2022,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute","funders":"","keywords":"Sample (material); Key (lock); Polynomial; Sample complexity; Robot; Swarm behaviour","score_opus":0.12786196221678384,"score_gpt":0.33449133675714904,"score_spread":0.2066293745403652,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124267618","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00846954,0.00014662705,0.98920244,0.00025067985,0.00002198899,0.000037263715,0.000054575405,0.00048855995,0.0013283285],"genre_scores_gemma":[0.7949906,0.00018568835,0.20131022,0.000263862,0.000059025962,0.00032960708,0.00022587983,0.00017509492,0.0024600436],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989843,0.00032891342,0.000044632827,0.00028381514,0.00023014603,0.00012806454],"domain_scores_gemma":[0.9969505,0.0017269321,0.0003491777,0.00046591295,0.00031985482,0.00018769201],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017598262,0.0010741835,0.0017714123,0.00039973497,0.0005443476,0.001295206,0.002044027,0.0013498919,0.0024397725],"category_scores_gemma":[0.0063978834,0.0007237901,0.00086189754,0.0005987547,0.0014853552,0.0018676282,0.0017558452,0.0031906762,0.00052069145],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000028845816,0.000025047368,0.00018075015,0.000025411107,0.000010676047,0.000018335884,0.000019685307,0.9860826,0.00029371987,0.0063440376,0.00040706072,0.006563778],"study_design_scores_gemma":[0.000008278294,0.0000096775875,0.000021821179,0.000002929794,0.0000019104284,0.0000039459223,0.0000028059928,0.9955214,0.00012544426,0.0042052926,0.00009417993,0.0000022539884],"about_ca_topic_score_codex":0.0050336486,"about_ca_topic_score_gemma":0.0052249343,"teacher_disagreement_score":0.0050336486,"about_ca_system_score_codex":0.0019424879,"about_ca_system_score_gemma":0.0029661579,"threshold_uncertainty_score":0.014093757},"labels":[],"label_agreement":null},{"id":"W7124272070","doi":"10.65109/emmn5327","title":"Neural Population Learning beyond Symmetric Zero-Sum Games","year":2024,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Google (Canada)","funders":"","keywords":"Convergence (economics); Suite; Population; Domain (mathematical analysis); Transfer of learning; Scale (ratio); Control (management); Stability (learning theory)","score_opus":0.017970437138648165,"score_gpt":0.26461794517087306,"score_spread":0.2466475080322249,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124272070","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07601224,0.0002545338,0.9138418,0.0006198656,0.000055824203,0.0000772249,0.00003064964,0.00016732392,0.008940591],"genre_scores_gemma":[0.9034896,0.00023467962,0.089900635,0.00027064685,0.000056693123,0.00023265847,0.000058631602,0.00008242887,0.0056740483],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99952245,0.00023379814,0.000016398682,0.00007477235,0.00008621138,0.000066312714],"domain_scores_gemma":[0.9970855,0.0022106743,0.0002046296,0.00014343289,0.00018236028,0.0001733958],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015994894,0.00078161556,0.0009772792,0.00039942097,0.00056785747,0.0011941446,0.0015858919,0.0011682871,0.0030764898],"category_scores_gemma":[0.008501643,0.00040102168,0.0005886287,0.00030036882,0.0020164493,0.0019760618,0.0022333167,0.0018498545,0.0003397177],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003669662,0.00005052984,0.00044489274,0.000046134617,0.000029620325,0.00007045545,0.000078970355,0.8745092,0.00080567197,0.115054026,0.00058347854,0.008290391],"study_design_scores_gemma":[0.0000075010785,0.000010915067,0.00002311081,0.0000029469147,0.0000016735141,0.000004518467,0.0000065491804,0.9673227,0.00008176048,0.032410443,0.00012587744,0.0000020658108],"about_ca_topic_score_codex":0.0035045627,"about_ca_topic_score_gemma":0.0031483923,"teacher_disagreement_score":0.0035045627,"about_ca_system_score_codex":0.0010921698,"about_ca_system_score_gemma":0.0010360113,"threshold_uncertainty_score":0.010291874},"labels":[],"label_agreement":null},{"id":"W7124275932","doi":"10.65109/dxvr3975","title":"Forward Actor-Critic for Nonlinear Function Approximation in Reinforcement Learning","year":2017,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Function approximation; Nonlinear system; Function (biology); Class (philosophy); Q-learning; Control theory (sociology)","score_opus":0.034713047450533524,"score_gpt":0.2929229350638085,"score_spread":0.258209887613275,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124275932","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0028818666,0.00038387673,0.9942247,0.00013749705,0.00005123269,0.000030393006,0.00001675512,0.0002939027,0.001979779],"genre_scores_gemma":[0.6641686,0.0007935123,0.32253128,0.00029501796,0.000104548366,0.00037837744,0.00011558641,0.00020659417,0.011406485],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99941945,0.00022027805,0.000031076404,0.00011703688,0.00016271803,0.000049440787],"domain_scores_gemma":[0.9981828,0.0012501132,0.00011314969,0.00014269099,0.0002537757,0.000057526493],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002076293,0.0012940938,0.0011580064,0.00041768193,0.00038483334,0.0008595822,0.0013494755,0.0014599026,0.0030717272],"category_scores_gemma":[0.0053134654,0.0006262467,0.00063319027,0.00045128696,0.0013389406,0.0008888181,0.0010389967,0.0027074534,0.00070139696],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000052483007,0.000031919677,0.0003257478,0.00009803284,0.00004042509,0.00007001235,0.000052730546,0.9315288,0.0011710168,0.028800014,0.0011649356,0.03666383],"study_design_scores_gemma":[0.0000050201993,0.000008881438,0.000018035898,0.000004204081,0.0000030325277,0.0000058042215,0.0000012136481,0.99542797,0.00020685686,0.003967514,0.00034861205,0.0000027398646],"about_ca_topic_score_codex":0.004950238,"about_ca_topic_score_gemma":0.0049609477,"teacher_disagreement_score":0.004950238,"about_ca_system_score_codex":0.0011130219,"about_ca_system_score_gemma":0.0012503315,"threshold_uncertainty_score":0.010980666},"labels":[],"label_agreement":null},{"id":"W7124283399","doi":"10.65109/cwyu2303","title":"Two-Level Actor-Critic Using Multiple Teachers","year":2023,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Alberta","funders":"","keywords":"Inefficiency; Variety (cybernetics); Sample (material); Domain (mathematical analysis); Action (physics); Forcing (mathematics)","score_opus":0.1396883853120315,"score_gpt":0.336168527520682,"score_spread":0.19648014220865048,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124283399","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011234146,0.00031654772,0.9793875,0.00042748128,0.00012848328,0.00011224512,0.000042777247,0.0012075616,0.0071432344],"genre_scores_gemma":[0.7753491,0.00020404825,0.2056591,0.00045368433,0.00013466831,0.0004512222,0.00016364122,0.00022415424,0.017360413],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99910897,0.0002480215,0.000049595594,0.0002703646,0.00020280974,0.00012023358],"domain_scores_gemma":[0.99770963,0.001375034,0.00015945388,0.00021325935,0.00032834403,0.00021420472],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001994855,0.0018226824,0.0018638266,0.0004993625,0.0006932816,0.0015194821,0.0034212018,0.0028382728,0.00839643],"category_scores_gemma":[0.004808561,0.0009426806,0.0006982407,0.00049205875,0.0017123871,0.0013439602,0.0024746058,0.0037060168,0.0019731375],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020279658,0.0001265689,0.0006559896,0.000105028535,0.00007629848,0.00018946812,0.000106344545,0.9437379,0.0015251642,0.016239796,0.0018607381,0.03517386],"study_design_scores_gemma":[0.000026393402,0.000023281278,0.00002328028,0.0000053721305,0.0000073192987,0.000011099258,0.000003978007,0.9969748,0.00022814513,0.0022866474,0.00040494435,0.000004745978],"about_ca_topic_score_codex":0.0049451343,"about_ca_topic_score_gemma":0.007304112,"teacher_disagreement_score":0.00839643,"about_ca_system_score_codex":0.0015069355,"about_ca_system_score_gemma":0.0016169932,"threshold_uncertainty_score":0.028088808},"labels":[],"label_agreement":null},{"id":"W7124285129","doi":"10.65109/vyhj2561","title":"Dual Ensembled Multiagent Q-Learning with Hypernet Regularizer","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Dual (grammatical number); Process (computing); Reinforcement learning; Computation; Multi-agent system","score_opus":0.009116201324229783,"score_gpt":0.2322789746532104,"score_spread":0.22316277332898063,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124285129","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015587178,0.00021045831,0.9824938,0.00019931495,0.000041312072,0.000046036075,0.00002120136,0.00018197467,0.0012187638],"genre_scores_gemma":[0.8335852,0.00019120735,0.16232117,0.00034873572,0.000091608876,0.00030258705,0.0001153277,0.00008976609,0.0029544209],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986507,0.0005093436,0.00008612174,0.0003351001,0.00026744982,0.00015131074],"domain_scores_gemma":[0.9962102,0.0020922872,0.00039236838,0.0003107288,0.0007994767,0.00019487107],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033449107,0.0012071553,0.0020927438,0.00065524806,0.0005758162,0.001115831,0.0021965934,0.0016572792,0.0019177545],"category_scores_gemma":[0.007566814,0.000664133,0.0006757987,0.0005588165,0.0013242334,0.0017255405,0.0022047302,0.0019923882,0.00034067966],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000081592494,0.00004933313,0.0010110635,0.000053072847,0.000061402454,0.000066371365,0.00006393443,0.95317197,0.0008388492,0.008776546,0.00067963084,0.035146277],"study_design_scores_gemma":[0.0000047395747,0.00001308671,0.000036089514,0.0000024985166,0.0000037061754,0.0000058537044,0.0000021331157,0.9981072,0.00012460357,0.0016147037,0.000083174156,0.0000022212732],"about_ca_topic_score_codex":0.0046776556,"about_ca_topic_score_gemma":0.004023141,"teacher_disagreement_score":0.0046776556,"about_ca_system_score_codex":0.0011070258,"about_ca_system_score_gemma":0.0016217498,"threshold_uncertainty_score":0.017689764},"labels":[],"label_agreement":null},{"id":"W7124289677","doi":"10.65109/ldoy7418","title":"Leveraging Sub-Optimal Data for Human-in-the-Loop Reinforcement Learning","year":2024,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Function (biology); Data collection; Human-in-the-loop; Temporal difference learning; Work (physics)","score_opus":0.09075955790988156,"score_gpt":0.3330438801573506,"score_spread":0.24228432224746904,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124289677","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022867415,0.00013999474,0.97434014,0.0002174854,0.000028687724,0.000046756024,0.000030181454,0.0005709326,0.0017583831],"genre_scores_gemma":[0.8201376,0.00009821528,0.17773683,0.00020746558,0.00002408552,0.00017205917,0.000104894774,0.00015034055,0.0013685706],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991912,0.00031154626,0.000041181425,0.00016117384,0.00020487761,0.00009002471],"domain_scores_gemma":[0.9961312,0.0024575123,0.00038047798,0.00044766493,0.000382573,0.00020058405],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021556786,0.0011139765,0.0011213511,0.00049103383,0.00045948743,0.0009877237,0.0012859036,0.0010663084,0.0024066658],"category_scores_gemma":[0.011232209,0.00062255707,0.00036181783,0.00033105508,0.0017192765,0.0016560969,0.0024922772,0.0024347433,0.0005367284],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012334262,0.00013474253,0.0010504948,0.000081750186,0.000028190181,0.000070636714,0.00011790723,0.9321096,0.0033598642,0.0082694385,0.00093227957,0.053721733],"study_design_scores_gemma":[0.000012159147,0.000045237193,0.000097249846,0.0000076731485,0.000002473871,0.000011150925,0.0000073009624,0.99410546,0.0010142988,0.0043299226,0.00036102152,0.0000060978327],"about_ca_topic_score_codex":0.0028252178,"about_ca_topic_score_gemma":0.004196095,"teacher_disagreement_score":0.0028252178,"about_ca_system_score_codex":0.0007283762,"about_ca_system_score_gemma":0.0017170935,"threshold_uncertainty_score":0.011400461},"labels":[],"label_agreement":null},{"id":"W7124291073","doi":"10.65109/yptr7088","title":"MaDi: Learning to Mask Distractions for Generalization in Visual Deep Reinforcement Learning","year":2024,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Generalization; Reinforcement learning; Focus (optics); Artificial neural network; Contrast (vision); Deep learning; Control (management)","score_opus":0.021834263229059014,"score_gpt":0.31080222842538047,"score_spread":0.28896796519632145,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124291073","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07641535,0.0007867812,0.91010046,0.00046591597,0.0001335781,0.00015511963,0.00013710516,0.008098305,0.0037074112],"genre_scores_gemma":[0.79834825,0.00015564731,0.1964888,0.00047126372,0.000048,0.00022365172,0.00030635172,0.00030898335,0.0036490909],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99963486,0.00007384043,0.000018834273,0.00010921882,0.00008953475,0.000073697745],"domain_scores_gemma":[0.9991866,0.00035249346,0.000113911905,0.00016157246,0.000099841476,0.00008569844],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014142912,0.0015140611,0.00092262187,0.00035251706,0.00033873765,0.0006563731,0.00232523,0.0011281843,0.0019521231],"category_scores_gemma":[0.003657062,0.00047354566,0.00050182035,0.00020239346,0.0009337825,0.001093805,0.0015812432,0.0023386446,0.00039807626],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002971031,0.00028147807,0.0022639162,0.00014928756,0.00011009077,0.00009155474,0.00014502493,0.7177654,0.013554949,0.0067771547,0.0056839664,0.25288007],"study_design_scores_gemma":[0.000024997198,0.000079250094,0.00013527114,0.0000081405315,0.000008640086,0.000015937962,0.000005783754,0.99397165,0.0030082983,0.0022654529,0.00047119314,0.000005339669],"about_ca_topic_score_codex":0.0045094676,"about_ca_topic_score_gemma":0.005379159,"teacher_disagreement_score":0.0045094676,"about_ca_system_score_codex":0.0012155627,"about_ca_system_score_gemma":0.001303212,"threshold_uncertainty_score":0.008966446},"labels":[],"label_agreement":null},{"id":"W7124291660","doi":"10.65109/mrqs8697","title":"Multiagent Q-learning with Sub-Team Coordination","year":2022,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada)","funders":"","keywords":"Exploit; Factorization; Class (philosophy); Monotonic function; Function (biology); Ranging","score_opus":0.012972007803506958,"score_gpt":0.22243070602490664,"score_spread":0.20945869822139968,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124291660","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0067729345,0.000098931356,0.991593,0.0001244956,0.000030395691,0.000040560208,0.000011924599,0.00013275584,0.0011949815],"genre_scores_gemma":[0.7793539,0.00014471123,0.21742587,0.000260326,0.00008206604,0.00026387218,0.0000721307,0.000054851545,0.0023422302],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99883157,0.0005035324,0.000050867722,0.000265759,0.00022455411,0.00012374356],"domain_scores_gemma":[0.9978738,0.0011345097,0.00026438863,0.00023043853,0.00029042145,0.00020650585],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026132888,0.0009156265,0.0012293538,0.00034312104,0.00049769453,0.00080752553,0.0019695584,0.0010331514,0.0026439177],"category_scores_gemma":[0.0054280036,0.0003824488,0.00047717767,0.00042615138,0.0011956004,0.0012618424,0.0018411287,0.0015556788,0.00049399637],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010341779,0.00012718324,0.00086022937,0.000095202544,0.000060840444,0.00007802046,0.00011513974,0.8857182,0.0016555244,0.03551375,0.0018966876,0.0737758],"study_design_scores_gemma":[0.000015590605,0.000035076388,0.00004027085,0.000003276616,0.000003427612,0.0000075818816,0.0000053356034,0.9904992,0.00019506566,0.008847818,0.0003444358,0.000002932916],"about_ca_topic_score_codex":0.0023603544,"about_ca_topic_score_gemma":0.0018890394,"teacher_disagreement_score":0.0026439177,"about_ca_system_score_codex":0.00070073415,"about_ca_system_score_gemma":0.0015656681,"threshold_uncertainty_score":0.013820589},"labels":[],"label_agreement":null},{"id":"W7124292505","doi":"10.65109/hdvm4783","title":"Monitored Markov Decision Processes","year":2024,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Markov decision process; Ask price; Partially observable Markov decision process; Formalism (music); Task (project management); Markov process","score_opus":0.01665031143990564,"score_gpt":0.2815242047084176,"score_spread":0.26487389326851196,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124292505","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011315034,0.0007065993,0.97886914,0.0009677703,0.00014074944,0.00007999206,0.00041499868,0.0003621919,0.007143578],"genre_scores_gemma":[0.77260995,0.0016046088,0.21154676,0.0005333891,0.00028898707,0.00046179027,0.00078104716,0.00009232856,0.012081058],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9983747,0.0007256209,0.00009237947,0.00042026403,0.0002242092,0.00016278622],"domain_scores_gemma":[0.99379987,0.004455753,0.0006342337,0.00047760786,0.00036821124,0.0002642231],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002009706,0.0012572183,0.0014229572,0.00062129274,0.0006299345,0.0018383756,0.002104402,0.0017488861,0.005928978],"category_scores_gemma":[0.008176247,0.00054703717,0.0009885926,0.0008437298,0.002076642,0.0025989767,0.0016835396,0.0026940473,0.0007925878],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000109524866,0.00007055519,0.001365757,0.00014646014,0.0000644966,0.00024769292,0.00014701486,0.34904826,0.0005970729,0.62769496,0.002491187,0.018017014],"study_design_scores_gemma":[0.00003573766,0.000028663537,0.00013300039,0.000019712808,0.000015003897,0.000036832524,0.000015712361,0.71134275,0.00017687655,0.28566352,0.0025187882,0.000013374377],"about_ca_topic_score_codex":0.003967456,"about_ca_topic_score_gemma":0.0047067595,"teacher_disagreement_score":0.005928978,"about_ca_system_score_codex":0.0018243854,"about_ca_system_score_gemma":0.0013538535,"threshold_uncertainty_score":0.01983434},"labels":[],"label_agreement":null},{"id":"W7124294995","doi":"10.65109/igaq8141","title":"Taming Multi-Agent Reinforcement Learning with Estimator Variance Reduction","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Estimator; Variance (accounting); Reinforcement learning; Variance reduction; Range (aeronautics); Sample (material); Reduction (mathematics)","score_opus":0.018782557438471854,"score_gpt":0.26857039620634454,"score_spread":0.2497878387678727,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124294995","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017222213,0.00007606282,0.98113316,0.0001227316,0.00001848464,0.000027733306,0.000007762314,0.0005597481,0.0008321087],"genre_scores_gemma":[0.85182816,0.00006043799,0.14676225,0.00015075509,0.0000339513,0.00013017397,0.0000246382,0.00009736924,0.0009121749],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976273,0.0012390524,0.00008075033,0.00031332584,0.0005600324,0.00017958846],"domain_scores_gemma":[0.98712885,0.008266361,0.0013348281,0.0021889496,0.0008060102,0.00027510055],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045606685,0.0007856367,0.0011116671,0.00034829267,0.0003649131,0.0007824766,0.001544683,0.0008052963,0.00086618314],"category_scores_gemma":[0.01888504,0.0004910502,0.0004278502,0.00026576617,0.0016408149,0.0011300994,0.002373534,0.0021566132,0.00027036754],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002061876,0.00012840776,0.0013355071,0.00008847651,0.00006189906,0.000088050256,0.00016544445,0.90791243,0.006528972,0.027768064,0.0006405411,0.055076055],"study_design_scores_gemma":[0.000019411533,0.00006274102,0.00008634109,0.0000050885137,0.000004612684,0.0000117858435,0.0000045338793,0.99322295,0.0014610119,0.004868963,0.00024768073,0.000004893622],"about_ca_topic_score_codex":0.0014232979,"about_ca_topic_score_gemma":0.0012235064,"teacher_disagreement_score":0.0045606685,"about_ca_system_score_codex":0.0007323341,"about_ca_system_score_gemma":0.001239496,"threshold_uncertainty_score":0.024119377},"labels":[],"label_agreement":null},{"id":"W7124298994","doi":"10.65109/nind4053","title":"Do As You Teach: A Multi-Teacher Approach to Self-Play in Deep Reinforcement Learning","year":2023,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Institut national de psychiatrie légale Philippe-Pinel","funders":"","keywords":"Reinforcement learning; Automation; Robot; State (computer science); Task (project management); Robotics; Robot learning; Train","score_opus":0.0385510444796588,"score_gpt":0.2879504930576487,"score_spread":0.2493994485779899,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124298994","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013981178,0.00034475658,0.9803816,0.00079422526,0.00007592536,0.000062657826,0.000042941225,0.0005277925,0.0037889192],"genre_scores_gemma":[0.81739056,0.0002723385,0.17354248,0.00036947773,0.000093225164,0.00024746972,0.00006585855,0.00016869775,0.007849826],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995055,0.00023631896,0.00001762597,0.00008976955,0.000086885826,0.00006393803],"domain_scores_gemma":[0.9990314,0.00053386675,0.00009607058,0.00009891225,0.00011173179,0.00012801343],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015872983,0.0007860584,0.00075324264,0.00029520778,0.00042864584,0.0008632944,0.0021155756,0.0013025866,0.004865607],"category_scores_gemma":[0.0033024256,0.00044228212,0.00033499798,0.00027808876,0.0010628054,0.0015546535,0.0019638883,0.0024751332,0.00055979786],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004050572,0.00051376666,0.001967387,0.0001868256,0.0001239807,0.0001673316,0.0005014251,0.66302264,0.0042741788,0.10713236,0.0073572104,0.21434787],"study_design_scores_gemma":[0.000016352014,0.000035058627,0.00004969353,0.0000070825727,0.0000059701283,0.000007907385,0.0000090916465,0.9821218,0.00032996762,0.016712366,0.00070082716,0.0000037849413],"about_ca_topic_score_codex":0.0028885205,"about_ca_topic_score_gemma":0.004200993,"teacher_disagreement_score":0.004865607,"about_ca_system_score_codex":0.0010854128,"about_ca_system_score_gemma":0.001002395,"threshold_uncertainty_score":0.016277134},"labels":[],"label_agreement":null},{"id":"W7124299647","doi":"10.65109/njvx2967","title":"Automatic Noise Filtering with Dynamic Sparse Training in Deep Reinforcement Learning","year":2023,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Noise (video); Focus (optics); Robot; Code (set theory); Training (meteorology); Transfer of learning; Deep learning","score_opus":0.028281085939947218,"score_gpt":0.2598315582196306,"score_spread":0.2315504722796834,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124299647","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023145556,0.00022116353,0.97364,0.0002462922,0.000042073152,0.00004352796,0.00004891251,0.0012354183,0.0013769908],"genre_scores_gemma":[0.81973314,0.00015004071,0.17724192,0.0002907675,0.00004847558,0.00018105617,0.00017736162,0.0001479984,0.0020292408],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995117,0.00014548356,0.000024250263,0.00011205544,0.00013132919,0.00007515538],"domain_scores_gemma":[0.99828464,0.001058756,0.00017348169,0.00017469317,0.00021706421,0.00009128087],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014325428,0.0008645164,0.0008766447,0.00029873927,0.00032708936,0.0005767387,0.001350529,0.00088842184,0.0015737563],"category_scores_gemma":[0.0057784487,0.000453787,0.00040127133,0.00031149335,0.0012226676,0.0010175538,0.001157966,0.0019629102,0.0003354926],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009785698,0.000083485946,0.00091019046,0.00006082276,0.000030892894,0.000045086235,0.000053454245,0.9259182,0.0026493901,0.008246348,0.0011668307,0.06073742],"study_design_scores_gemma":[0.000009255424,0.000017362629,0.000035770216,0.0000034993634,0.0000023839066,0.000004703388,0.000002115092,0.9965668,0.00046138905,0.002727761,0.00016659898,0.000002291833],"about_ca_topic_score_codex":0.0065391655,"about_ca_topic_score_gemma":0.0077381846,"teacher_disagreement_score":0.0065391655,"about_ca_system_score_codex":0.00095490174,"about_ca_system_score_gemma":0.00145602,"threshold_uncertainty_score":0.013002217},"labels":[],"label_agreement":null},{"id":"W7124301226","doi":"10.65109/reoc9270","title":"Boosting Robustness in Preference-Based Reinforcement Learning with Dynamic Sparsity","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Boosting (machine learning); Robustness (evolution); Focus (optics); Artificial neural network; Training set","score_opus":0.029300188627362417,"score_gpt":0.24804607384247793,"score_spread":0.2187458852151155,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124301226","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08854472,0.000331498,0.90720314,0.00041661054,0.000049624083,0.000074727315,0.00005181702,0.0007230587,0.0026048415],"genre_scores_gemma":[0.95326334,0.00009415249,0.04503148,0.00023148209,0.000035788576,0.000084682506,0.00006384647,0.000067663044,0.0011275191],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99893445,0.00041760484,0.00005098555,0.00021836007,0.00023933814,0.00013924105],"domain_scores_gemma":[0.99438596,0.0037023704,0.00067818427,0.00042616235,0.00053399825,0.00027332042],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027406446,0.0010187911,0.0012291501,0.00047509244,0.00038406265,0.0005773473,0.0014627081,0.0009170736,0.0012966786],"category_scores_gemma":[0.014130117,0.00051861256,0.00043360543,0.00030307713,0.001376178,0.0013234534,0.0015639858,0.0016715451,0.00027060168],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018215286,0.00010722786,0.0012122812,0.00006794433,0.000052504118,0.00006165082,0.000064141925,0.9517629,0.002742014,0.008987189,0.0007607297,0.033999212],"study_design_scores_gemma":[0.000015174176,0.00005192189,0.00007984147,0.000003702565,0.0000038332787,0.000011410371,0.000003177127,0.9954085,0.00032452546,0.00400566,0.000088210494,0.0000039898355],"about_ca_topic_score_codex":0.0035805851,"about_ca_topic_score_gemma":0.0030930704,"teacher_disagreement_score":0.0035805851,"about_ca_system_score_codex":0.0009917826,"about_ca_system_score_gemma":0.0010538492,"threshold_uncertainty_score":0.014494121},"labels":[],"label_agreement":null},{"id":"W7124301997","doi":"10.65109/tjqd1268","title":"PORTAL: Automatic Curricula Generation for Multiagent Reinforcement Learning","year":2023,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Task (project management); Curriculum; Feature (linguistics); Space (punctuation); Active learning (machine learning); Key (lock); Transfer of learning","score_opus":0.05505160572722048,"score_gpt":0.3059342340578389,"score_spread":0.2508826283306185,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124301997","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0077179074,0.00009536284,0.98721886,0.00009888394,0.00003806575,0.00010142022,0.000052592542,0.0036504196,0.0010264716],"genre_scores_gemma":[0.46020058,0.00012332214,0.53529257,0.0001794953,0.000045641646,0.0006918488,0.00032643293,0.0003793539,0.0027607353],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994554,0.00021744255,0.00002869961,0.00011976866,0.00011635109,0.00006225871],"domain_scores_gemma":[0.99887425,0.00056703726,0.00011734062,0.0001443024,0.00017762417,0.00011949463],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016553585,0.00086200476,0.00089860783,0.0004917594,0.00041007812,0.0006718481,0.001908493,0.0010159643,0.005561862],"category_scores_gemma":[0.004514312,0.00046237258,0.00048577334,0.00028266237,0.0007299244,0.0010295905,0.0018241995,0.0016347535,0.0011329904],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023150117,0.00028883087,0.0017364892,0.00019695013,0.00006694829,0.00012753741,0.0001431795,0.7384765,0.0040671513,0.017559081,0.0059211254,0.23118478],"study_design_scores_gemma":[0.00003266064,0.000028838916,0.00005438908,0.0000043496843,0.0000029444798,0.000008652566,0.000004466426,0.9952565,0.00055436767,0.0032975355,0.0007517548,0.0000034613383],"about_ca_topic_score_codex":0.0030350839,"about_ca_topic_score_gemma":0.0031661368,"teacher_disagreement_score":0.005561862,"about_ca_system_score_codex":0.00084568927,"about_ca_system_score_gemma":0.0015415407,"threshold_uncertainty_score":0.018606246},"labels":[],"label_agreement":null},{"id":"W7124304825","doi":"10.65109/vzns8734","title":"A Model-Based Solution to the Offline Multi-Agent Reinforcement Learning Coordination Problem","year":2024,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Reinforcement learning; Leverage (statistics); Observability; Online and offline; Simple (philosophy); Offline learning; Online algorithm","score_opus":0.04538121631456204,"score_gpt":0.2906629308777778,"score_spread":0.24528171456321574,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124304825","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005716526,0.00005813141,0.99135196,0.00019483722,0.000023194922,0.00004178069,0.00003310971,0.0002516483,0.0023287884],"genre_scores_gemma":[0.7022082,0.00011646799,0.29207864,0.0001952508,0.000052553747,0.00030287966,0.00014600587,0.00014393363,0.004756158],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994393,0.00018866999,0.000021512673,0.00015843833,0.000115999115,0.00007598808],"domain_scores_gemma":[0.99853396,0.0007886937,0.00019928714,0.00018263077,0.00015470048,0.00014073233],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011059284,0.001067912,0.0013976543,0.00032659923,0.0004839157,0.000884809,0.0015851732,0.0013759925,0.0030773615],"category_scores_gemma":[0.0034023116,0.00056064955,0.00057104835,0.00029569256,0.0011353234,0.00090076716,0.0016704545,0.002163377,0.0005816305],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000030051066,0.00003153914,0.00020524643,0.000040104384,0.0000145244985,0.000043558477,0.000033857476,0.9757043,0.0006304328,0.009365603,0.0007491882,0.013151666],"study_design_scores_gemma":[0.000009200754,0.000019206163,0.00003063247,0.0000034760906,0.0000023205714,0.000010088306,0.0000061554656,0.99501026,0.00015983071,0.004444549,0.00030188274,0.0000023715752],"about_ca_topic_score_codex":0.0037399887,"about_ca_topic_score_gemma":0.0034268554,"teacher_disagreement_score":0.0037399887,"about_ca_system_score_codex":0.0007460752,"about_ca_system_score_gemma":0.0020296094,"threshold_uncertainty_score":0.010294795},"labels":[],"label_agreement":null},{"id":"W7124307291","doi":"10.65109/qtfy6777","title":"The Importance of Credo in Multiagent Learning","year":2023,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Context (archaeology); Population; Multi-agent system; Social learning","score_opus":0.03153244086601935,"score_gpt":0.2863773328859435,"score_spread":0.25484489201992416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124307291","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07565485,0.0007308654,0.9093652,0.003096444,0.00017604728,0.00008331236,0.00008610976,0.00018582953,0.010621461],"genre_scores_gemma":[0.96265894,0.00021181937,0.035166223,0.00025684663,0.00009506398,0.00008685326,0.000027475728,0.00003250828,0.001464284],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99532795,0.0028425278,0.00013634287,0.0005598046,0.0007418502,0.00039161227],"domain_scores_gemma":[0.97182614,0.020926684,0.0015347016,0.0024424773,0.0012224312,0.0020475825],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068642166,0.0008863047,0.0016824459,0.000597593,0.0012738906,0.0032668065,0.0021759432,0.0024312332,0.0024962528],"category_scores_gemma":[0.032042712,0.00061531435,0.00071921933,0.000659659,0.005102304,0.005600527,0.0032617578,0.0042622136,0.00028115403],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001910825,0.00013131078,0.001805632,0.00009075521,0.000054803088,0.00014444489,0.00017599248,0.6651253,0.0008300783,0.31631494,0.0009892023,0.014146466],"study_design_scores_gemma":[0.000020619254,0.00004808209,0.000111367925,0.00001247374,0.0000057921766,0.000023312194,0.000016738046,0.84153813,0.00014626897,0.15770347,0.00036031488,0.000013438117],"about_ca_topic_score_codex":0.0022355022,"about_ca_topic_score_gemma":0.0017464585,"teacher_disagreement_score":0.0068642166,"about_ca_system_score_codex":0.001970091,"about_ca_system_score_gemma":0.0022193517,"threshold_uncertainty_score":0.03630191},"labels":[],"label_agreement":null},{"id":"W7124309298","doi":"10.65109/vlwf1223","title":"Transfer Learning based Agent for Automated Negotiation","year":2023,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Universiteit Maastricht; National Natural Science Foundation of China","keywords":"Negotiation; Transfer of learning; Adaptation (eye); Context (archaeology); Knowledge transfer; Key (lock); Task (project management); Boosting (machine learning)","score_opus":0.03994887653339272,"score_gpt":0.2890886402997899,"score_spread":0.2491397637663972,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124309298","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011124979,0.000291182,0.97923917,0.00031254347,0.000079667385,0.0001557949,0.000026600848,0.0011483153,0.0076218206],"genre_scores_gemma":[0.7299735,0.00032314434,0.2596667,0.00023178579,0.00008427348,0.0005863339,0.00009436239,0.00011850615,0.008921408],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99911064,0.0003307134,0.00004810867,0.00014235014,0.00028056014,0.000087576525],"domain_scores_gemma":[0.9990369,0.00049298623,0.00009846518,0.00013786666,0.00014975836,0.000084016276],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015380137,0.0006738424,0.0007367465,0.0004643412,0.0007083218,0.00090178574,0.0017833327,0.0013962174,0.0060874415],"category_scores_gemma":[0.003290151,0.00027211753,0.0005343764,0.00044377422,0.0013137027,0.0020165087,0.0020438577,0.002090944,0.0011909317],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024616695,0.000321472,0.00090316025,0.00020432474,0.00009728737,0.00035769565,0.00027960527,0.66701275,0.008517332,0.10352736,0.0050631715,0.2134697],"study_design_scores_gemma":[0.000017839455,0.00003630665,0.0000538655,0.0000059546874,0.0000050295457,0.000033740875,0.000011961068,0.9754361,0.0011775926,0.021006893,0.0022067137,0.000007967272],"about_ca_topic_score_codex":0.0017685908,"about_ca_topic_score_gemma":0.0011441926,"teacher_disagreement_score":0.0060874415,"about_ca_system_score_codex":0.0009163521,"about_ca_system_score_gemma":0.0012831411,"threshold_uncertainty_score":0.020364583},"labels":[],"label_agreement":null},{"id":"W7124313799","doi":"10.65109/vmil8420","title":"Off-the-Grid MARL: Datasets and Baselines for Offline Multi-Agent Reinforcement Learning","year":2023,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institut National de la Recherche Scientifique","funders":"","keywords":"Reinforcement learning; Bespoke; Online and offline; Measure (data warehouse); Power (physics)","score_opus":0.05639460265554716,"score_gpt":0.3100390074822671,"score_spread":0.2536444048267199,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124313799","genre_codex":"methods","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08855102,0.0063095693,0.70184505,0.006235448,0.0021972936,0.0020070432,0.05546823,0.110465944,0.026920367],"genre_scores_gemma":[0.33930114,0.0011142322,0.5276157,0.0014943671,0.00026551783,0.0022893446,0.117288366,0.006467701,0.004163658],"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","domain_scores_codex":[0.99283653,0.003333983,0.00046539406,0.001312064,0.0016452122,0.00040680828],"domain_scores_gemma":[0.9709036,0.010020122,0.0012373045,0.012929703,0.0037871955,0.0011221191],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011433317,0.0024862054,0.0016943152,0.0024725243,0.0012056637,0.0034788274,0.007778214,0.003320813,0.0068982895],"category_scores_gemma":[0.062404837,0.0011269802,0.001796381,0.002428249,0.002417879,0.005981507,0.0058065685,0.005259415,0.006048348],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022020515,0.0023014992,0.0089459205,0.0018540056,0.0004917233,0.0003291199,0.00025893844,0.54640156,0.0025133905,0.028945781,0.17442796,0.23132804],"study_design_scores_gemma":[0.00034142382,0.00038101984,0.0016614606,0.00019430937,0.000041150102,0.00013407953,0.0001195117,0.92889786,0.00341544,0.035866015,0.028880782,0.000066976085],"about_ca_topic_score_codex":0.0067579853,"about_ca_topic_score_gemma":0.011158983,"teacher_disagreement_score":0.011433317,"about_ca_system_score_codex":0.0022274135,"about_ca_system_score_gemma":0.002829957,"threshold_uncertainty_score":0.060465872},"labels":[],"label_agreement":null},{"id":"W7124317253","doi":"10.65109/gzci2494","title":"Value Iteration for Learning Concurrently Executable Robotic Control Tasks","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Executable; Task (project management); Reinforcement learning; Set (abstract data type); Independence (probability theory); Robot; Property (philosophy); Control (management)","score_opus":0.013691952864955292,"score_gpt":0.2791122585949014,"score_spread":0.2654203057299461,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124317253","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022820476,0.0001528053,0.97505414,0.00016444443,0.00002328802,0.000054145024,0.0000141096025,0.0002177681,0.0014988566],"genre_scores_gemma":[0.84022087,0.00012966838,0.15635484,0.00012612317,0.00003052955,0.00029165554,0.0000763735,0.00011019117,0.002659679],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99892884,0.0004893397,0.000048865113,0.0001582324,0.000248194,0.00012659306],"domain_scores_gemma":[0.99630934,0.002809702,0.0002394847,0.00014293018,0.00035320123,0.00014538356],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029044873,0.0012713658,0.001312152,0.0005924674,0.00037595793,0.0007377185,0.0012734241,0.0014529501,0.0018518893],"category_scores_gemma":[0.00810316,0.0007945593,0.0006406671,0.00044318318,0.0018569361,0.0013116254,0.0014237609,0.002183803,0.00028254327],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005263985,0.000027064803,0.00028223268,0.000027975804,0.000018778215,0.00002353293,0.00003671099,0.9806231,0.0005357564,0.007487492,0.00016665789,0.0107179545],"study_design_scores_gemma":[0.000005597705,0.000014966983,0.000014227463,0.0000022839024,0.0000011683904,0.0000022624365,0.0000018008753,0.99715364,0.00014524197,0.002601388,0.00005580693,0.000001684073],"about_ca_topic_score_codex":0.004081519,"about_ca_topic_score_gemma":0.0030662564,"teacher_disagreement_score":0.004081519,"about_ca_system_score_codex":0.0015033638,"about_ca_system_score_gemma":0.0014700035,"threshold_uncertainty_score":0.015360594},"labels":[],"label_agreement":null},{"id":"W7124317441","doi":"10.65109/ltgs2519","title":"Be Considerate: Avoiding Negative Side Effects in Reinforcement Learning","year":2022,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute","funders":"","keywords":"Reinforcement learning; Agency (philosophy); Reinforcement; Control (management); Action (physics); Discretion","score_opus":0.02380526822181108,"score_gpt":0.25947887441085216,"score_spread":0.23567360618904107,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124317441","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15800837,0.00040449086,0.8213705,0.0014828447,0.00008279024,0.00021127706,0.000048330603,0.0011390175,0.017252328],"genre_scores_gemma":[0.926269,0.000087187684,0.070775546,0.00028402515,0.000019851375,0.00015835857,0.000028175586,0.00009136737,0.002286463],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99757725,0.0013242728,0.00009484133,0.0003742825,0.00044536206,0.00018400101],"domain_scores_gemma":[0.99135816,0.0052376837,0.0010349732,0.0012626254,0.0006206943,0.0004858599],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045659016,0.0009883732,0.00073785277,0.0002915151,0.0007974113,0.0012477308,0.0011774645,0.0012867581,0.0031459644],"category_scores_gemma":[0.017024605,0.0003476128,0.0003780909,0.00024791528,0.0025326936,0.0022172737,0.0021978072,0.0021096778,0.00050937204],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015620198,0.0008216709,0.013157615,0.00046212203,0.00023350517,0.0006074827,0.0014923245,0.5372607,0.02300439,0.18569395,0.003143401,0.2325608],"study_design_scores_gemma":[0.00021491431,0.0007954231,0.0013470444,0.00006476412,0.00007395491,0.00019336691,0.000122324,0.81458014,0.005769691,0.17305265,0.003746141,0.00003960197],"about_ca_topic_score_codex":0.0011866569,"about_ca_topic_score_gemma":0.0015510367,"teacher_disagreement_score":0.0045659016,"about_ca_system_score_codex":0.0006987636,"about_ca_system_score_gemma":0.00126233,"threshold_uncertainty_score":0.024147034},"labels":[],"label_agreement":null},{"id":"W7124320264","doi":"10.65109/jazo1666","title":"A Deeper Look at Discounting Mismatch in Actor-Critic Algorithms","year":2022,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Microsoft (Canada)","funders":"","keywords":"Discounting; Perspective (graphical); Representation (politics); Term (time); Trajectory; Task (project management); Value (mathematics); Empirical research","score_opus":0.019374595721380002,"score_gpt":0.25472915205792435,"score_spread":0.23535455633654434,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124320264","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020220596,0.0010410531,0.9728387,0.0015210377,0.00013194955,0.00003910276,0.000021953683,0.00019046199,0.003995153],"genre_scores_gemma":[0.7865157,0.00089607335,0.20705453,0.00063868816,0.00022488782,0.000111970134,0.000048467788,0.00029251596,0.004217025],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99632806,0.0017632204,0.0002166239,0.000671584,0.00078157824,0.00023886078],"domain_scores_gemma":[0.98696554,0.009655801,0.0009275967,0.001070899,0.0009411781,0.00043898166],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0094585745,0.0014258288,0.0017036208,0.0005288871,0.000569667,0.0030720427,0.0027769592,0.0025844038,0.0035432004],"category_scores_gemma":[0.04405364,0.00089051394,0.0008161795,0.0006522251,0.0020871349,0.007187878,0.0027440563,0.005931761,0.00037763233],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027170786,0.00012978954,0.0024244327,0.00025299477,0.00017355985,0.00017298365,0.0005007355,0.62083834,0.0032798569,0.32023275,0.0013983701,0.05032448],"study_design_scores_gemma":[0.000024322922,0.00007708688,0.00016464943,0.000035931458,0.000021841886,0.000044290922,0.000021833139,0.9127333,0.001056184,0.0844932,0.0013065138,0.00002088003],"about_ca_topic_score_codex":0.0023368238,"about_ca_topic_score_gemma":0.0016406954,"teacher_disagreement_score":0.0094585745,"about_ca_system_score_codex":0.002361526,"about_ca_system_score_gemma":0.0014427884,"threshold_uncertainty_score":0.050022304},"labels":[],"label_agreement":null},{"id":"W7124320494","doi":"10.65109/vaba4866","title":"PADDLE: Logic Program Guided Policy Reuse in Deep Reinforcement Learning","year":2024,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reuse; Reinforcement learning; Metric (unit); Transferability; Task (project management); Similarity (geometry); Measure (data warehouse)","score_opus":0.04022714982400774,"score_gpt":0.33431780415230783,"score_spread":0.29409065432830006,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124320494","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025915196,0.00023976857,0.97005266,0.0002183532,0.000035114346,0.0001013988,0.000051069543,0.0019241847,0.0014622211],"genre_scores_gemma":[0.8298859,0.00016283648,0.16627413,0.0003472406,0.000030051742,0.00026731056,0.0001548201,0.00019053789,0.0026871318],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99911934,0.00028287258,0.00004786083,0.00022509444,0.00019580673,0.00012905344],"domain_scores_gemma":[0.9985183,0.0008156074,0.00017951023,0.00019360139,0.00015877961,0.00013423235],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017601949,0.00092652073,0.0008975321,0.0004346595,0.00031073194,0.00075398234,0.0018512063,0.00103584,0.0026429247],"category_scores_gemma":[0.005210537,0.00042679964,0.0004949296,0.00032203438,0.001138392,0.0017002918,0.0016416447,0.0018502094,0.00038872924],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026811368,0.0003032979,0.0021702023,0.00018304859,0.00008255785,0.00011750613,0.0001568232,0.73482436,0.005565677,0.0156221595,0.0023369652,0.23836932],"study_design_scores_gemma":[0.00001966753,0.000056949953,0.00006901112,0.0000071025133,0.0000068676386,0.0000102549375,0.0000051564225,0.9921903,0.0011899124,0.0060920636,0.00034751408,0.0000051496063],"about_ca_topic_score_codex":0.003919579,"about_ca_topic_score_gemma":0.004027668,"teacher_disagreement_score":0.003919579,"about_ca_system_score_codex":0.0012094837,"about_ca_system_score_gemma":0.0017902162,"threshold_uncertainty_score":0.009308934},"labels":[],"label_agreement":null},{"id":"W7124326325","doi":"10.65109/sjgw9760","title":"Empowering Generalization for Deep Reinforcement Learning via Symbolic Planning","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Generalization; Pearl; Planner; Limiting; Automated planning and scheduling; Plan (archaeology); Deep learning","score_opus":0.019387053090284885,"score_gpt":0.30838076534145553,"score_spread":0.28899371225117065,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124326325","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.054176442,0.0004553464,0.9391499,0.00052070327,0.00004319304,0.000068979956,0.00011372602,0.002201695,0.0032699709],"genre_scores_gemma":[0.8668857,0.00021409879,0.13067754,0.00018870183,0.000021434791,0.00015075142,0.00019316682,0.00010740848,0.0015611807],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995921,0.00013779353,0.000022511622,0.00011449388,0.00007657372,0.00005642709],"domain_scores_gemma":[0.99845326,0.00096594554,0.00012138565,0.00027371224,0.00010991577,0.00007585652],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011841338,0.0009145749,0.0007665437,0.0003496797,0.00032625752,0.0005699902,0.001258947,0.000789472,0.0023728262],"category_scores_gemma":[0.004902762,0.00042337045,0.00052120193,0.00033530858,0.0013493623,0.0015925082,0.0015630069,0.0022319208,0.0003302623],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000084879095,0.00007594304,0.0010039962,0.00009265579,0.000036129695,0.00006600009,0.000083924235,0.8943962,0.0021196334,0.012115253,0.001391593,0.088533744],"study_design_scores_gemma":[0.000009894693,0.00002186673,0.000053757103,0.000005874602,0.0000033735898,0.000006105602,0.000004424991,0.99111146,0.0004296807,0.008084093,0.0002666168,0.0000028354611],"about_ca_topic_score_codex":0.005029621,"about_ca_topic_score_gemma":0.007883764,"teacher_disagreement_score":0.005029621,"about_ca_system_score_codex":0.0009684565,"about_ca_system_score_gemma":0.0013797461,"threshold_uncertainty_score":0.010000706},"labels":[],"label_agreement":null},{"id":"W7125578405","doi":"10.1109/mecatronics-rem67547.2025.11349467","title":"Learning Multistage Robotic Manipulation Using Chained Options and Composable Subtask Rewards","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Interpretability; Task (project management); Benchmark (surveying); Stability (learning theory); Decomposition; Sequence (biology); Process (computing)","score_opus":0.031009359942283158,"score_gpt":0.29120584357066337,"score_spread":0.2601964836283802,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7125578405","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11959953,0.00014180502,0.87776953,0.00019263719,0.000020004762,0.00006515635,0.00003889495,0.00045295438,0.0017194201],"genre_scores_gemma":[0.933774,0.00005214471,0.064540565,0.000054997377,0.000007457473,0.000104819446,0.00004493381,0.000039407918,0.001381696],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996037,0.00013726446,0.000021546439,0.00009775322,0.00007455178,0.00006518354],"domain_scores_gemma":[0.9987212,0.00079814397,0.00013848436,0.00012447503,0.0000960598,0.00012162497],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011114278,0.00078165013,0.0006786469,0.00028429777,0.00031762294,0.0005483426,0.0010744756,0.0008850785,0.0016882778],"category_scores_gemma":[0.003256041,0.00051529147,0.000513519,0.00020088001,0.0011782398,0.0012201982,0.0012447591,0.001428058,0.0001922933],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007554027,0.00003888005,0.00066869566,0.000023853003,0.000017131344,0.000048934646,0.00004816882,0.9730584,0.0021535258,0.005163681,0.00014906378,0.018554192],"study_design_scores_gemma":[0.0000056854124,0.000017299004,0.00004891072,0.000002229412,0.0000018545278,0.0000033721597,0.0000025463453,0.99615306,0.00036109504,0.0033503529,0.000051525352,0.0000021132926],"about_ca_topic_score_codex":0.0029668293,"about_ca_topic_score_gemma":0.0043606427,"teacher_disagreement_score":0.0029668293,"about_ca_system_score_codex":0.00084349787,"about_ca_system_score_gemma":0.00096701866,"threshold_uncertainty_score":0.006120026},"labels":[],"label_agreement":null},{"id":"W7125650556","doi":"10.1109/cascon66301.2025.00079","title":"Supervised Semantic Similarity-Based Conflict Detection Algorithm: S3CDA","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"IBM (Canada); Toronto Metropolitan University","funders":"","keywords":"Pattern recognition (psychology); Feature (linguistics); Noise (video); Semantics (computer science)","score_opus":0.02035180397387599,"score_gpt":0.2674203222397192,"score_spread":0.2470685182658432,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7125650556","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037208512,0.00030154746,0.95487636,0.00026017736,0.00015796123,0.00031210567,0.00020654062,0.0032264297,0.0034504344],"genre_scores_gemma":[0.457416,0.00009054449,0.53645647,0.00034339805,0.000059283208,0.00037528508,0.00069256924,0.00019455214,0.0043719225],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990884,0.00013513977,0.0000597813,0.00030080415,0.00031078514,0.000105121166],"domain_scores_gemma":[0.99871385,0.0003256658,0.000113453716,0.00018257988,0.0005413015,0.00012308909],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010195718,0.00083746493,0.0016334289,0.0014303896,0.00094864564,0.001043798,0.0030683107,0.0015236549,0.0057012616],"category_scores_gemma":[0.0026884468,0.00039077929,0.00073327654,0.001006881,0.0007571969,0.0010514676,0.0020570415,0.0014149958,0.0011969564],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005890329,0.00073889457,0.0037135724,0.00014085571,0.00015465076,0.00013411454,0.00010149965,0.11796083,0.01360005,0.007531357,0.011811555,0.84352356],"study_design_scores_gemma":[0.000051382103,0.00008112995,0.00042580342,0.000007288018,0.000019445337,0.000071748334,0.000022399725,0.990555,0.0041703857,0.0032099457,0.001371842,0.000013778189],"about_ca_topic_score_codex":0.007082485,"about_ca_topic_score_gemma":0.009638954,"teacher_disagreement_score":0.007082485,"about_ca_system_score_codex":0.0008491073,"about_ca_system_score_gemma":0.0033285618,"threshold_uncertainty_score":0.019072652},"labels":[],"label_agreement":null},{"id":"W7125958253","doi":"10.1145/3768292.3793403","title":"10.1145/3768292.3793403","year":2000,"lang":"en","type":"article","venue":"Time to knit","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Session (web analytics); Reinforcement; Reinforcement learning; Component (thermodynamics); Training (meteorology)","score_opus":0.006230724881324144,"score_gpt":0.179140084893979,"score_spread":0.17290936001265486,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7125958253","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004110769,0.011117388,0.026504349,0.0021930027,0.0030280892,0.0007928998,0.026896486,0.030857138,0.89449996],"genre_scores_gemma":[0.006108896,0.0036041352,0.003817214,0.00097036833,0.00017197203,0.00033004227,0.011377981,0.0028078686,0.9708114],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99933213,0.00005020454,0.00005978545,0.0002064888,0.00021798209,0.00013328626],"domain_scores_gemma":[0.998569,0.0003571822,0.000084562715,0.0004910874,0.00027371783,0.00022448436],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0018446477,0.0041000927,0.00300509,0.0025435572,0.0020132246,0.004098115,0.0033000123,0.006074962,0.9232635],"category_scores_gemma":[0.0028366323,0.0021767956,0.0014913481,0.008406305,0.0014965615,0.009554408,0.00594667,0.0035350227,0.9436111],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036774809,0.00025556807,0.00052920484,0.00078203797,0.00006544127,0.0002612014,0.0000798589,0.0014547535,0.0017023492,0.005897945,0.6175505,0.37105343],"study_design_scores_gemma":[0.000064154694,0.000048391554,0.00087080174,0.00034547952,0.00006430268,0.00019425378,0.000063544496,0.0019444422,0.00082819106,0.0025816201,0.9929531,0.00004160837],"about_ca_topic_score_codex":0.017541438,"about_ca_topic_score_gemma":0.01229592,"teacher_disagreement_score":0.07673651,"about_ca_system_score_codex":0.0021144485,"about_ca_system_score_gemma":0.0010347501,"threshold_uncertainty_score":0.10945535},"labels":[],"label_agreement":null},{"id":"W7131116789","doi":"10.1109/robio66223.2025.11378377","title":"A Strategy Adaptive Adjustment Deep Reinforcement Learning Method with Behavior Cloning for Mobile Robot Navigation","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Reinforcement learning; Obstacle avoidance; Randomness; Obstacle; Robot; Mobile robot; Trajectory; Training (meteorology); Path (computing)","score_opus":0.025411914609538396,"score_gpt":0.3187286794351319,"score_spread":0.2933167648255935,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7131116789","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020859582,0.0003723311,0.9757607,0.00018497847,0.00008001387,0.000049275946,0.00002636266,0.001063499,0.0016031998],"genre_scores_gemma":[0.8275054,0.00022939744,0.16658378,0.00031703882,0.00004267748,0.00022136298,0.00013028595,0.0001174824,0.004852483],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997609,0.000051209765,0.000015047505,0.000060899634,0.00006574435,0.000046105735],"domain_scores_gemma":[0.9996438,0.00012686163,0.000049116952,0.00003375833,0.00010340733,0.00004307938],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005907048,0.00079978554,0.0008631829,0.0003131193,0.00029729278,0.00041407684,0.0013535718,0.0006890837,0.0016645287],"category_scores_gemma":[0.0013539477,0.00039888898,0.00056609174,0.00024142815,0.0004749951,0.0005544413,0.0008756567,0.0011686494,0.00030174656],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012452381,0.00011347575,0.0015994735,0.00008736223,0.00007221326,0.00013960045,0.00011151997,0.7865726,0.0071915654,0.006198813,0.0024838932,0.19530496],"study_design_scores_gemma":[0.000008975206,0.000023298546,0.000050471022,0.000002623317,0.0000045568795,0.000008895038,0.00000203121,0.99876815,0.0003816482,0.000509452,0.00023686749,0.0000029626312],"about_ca_topic_score_codex":0.008207185,"about_ca_topic_score_gemma":0.005758546,"teacher_disagreement_score":0.008207185,"about_ca_system_score_codex":0.00066688284,"about_ca_system_score_gemma":0.0013870986,"threshold_uncertainty_score":0.016318798},"labels":[],"label_agreement":null},{"id":"W7132054910","doi":"","title":"Time and temporal abstraction in continual learning: tradeoffs, analogies and regret in an active measuring setting","year":2023,"lang":"en","type":"article","venue":"NPARC","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Abstraction; Regret; Context (archaeology); Class (philosophy); Markov decision process; Relevance (law); Hierarchy; Reinforcement learning; Simple (philosophy)","score_opus":0.03017605342244424,"score_gpt":0.263177105349338,"score_spread":0.2330010519268938,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7132054910","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.060503487,0.0009301616,0.9264378,0.0022468164,0.00007627968,0.000051138988,0.00008029801,0.00021948916,0.009454491],"genre_scores_gemma":[0.90965873,0.00050572964,0.08657747,0.00021852729,0.00013716178,0.00012616561,0.00007349895,0.00007455366,0.0026281497],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.994955,0.002383363,0.00018929424,0.0009013747,0.0011427787,0.0004281662],"domain_scores_gemma":[0.96753514,0.025394063,0.002093024,0.0029132762,0.00079712324,0.0012673493],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00805938,0.0010424212,0.0013352069,0.00074429525,0.0008789142,0.0037391235,0.0025026752,0.002314603,0.004415663],"category_scores_gemma":[0.038621385,0.0006871858,0.0012450176,0.0008980463,0.00563566,0.010335027,0.0046193865,0.005702789,0.000330856],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020133232,0.00014288246,0.0010758066,0.00015294779,0.00006306294,0.000101314195,0.00036128258,0.22425811,0.00074516743,0.743146,0.0006538748,0.029098153],"study_design_scores_gemma":[0.000019243342,0.000083390456,0.0002404653,0.000027422337,0.000012792463,0.000027949704,0.00003578198,0.42054325,0.00023922249,0.57816625,0.0005871211,0.000017134491],"about_ca_topic_score_codex":0.00151191,"about_ca_topic_score_gemma":0.0011142249,"teacher_disagreement_score":0.00805938,"about_ca_system_score_codex":0.0029944698,"about_ca_system_score_gemma":0.0012517179,"threshold_uncertainty_score":0.042622626},"labels":[],"label_agreement":null},{"id":"W7132865739","doi":"","title":"Constrained, Curiosity-driven Trajectory Optimization for Learned Quadrotor Control","year":2022,"lang":"","type":"dissertation","venue":"TSpace","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Trajectory; Stability (learning theory); Probabilistic logic; Reinforcement learning; Control theory (sociology); Robot; Control (management); Model predictive control; Optimal control; State space","score_opus":0.029156504398305867,"score_gpt":0.33801370815472515,"score_spread":0.3088572037564193,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7132865739","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030848319,0.00014210826,0.9653208,0.00023977763,0.000025526131,0.000034814202,0.000030317395,0.00018777317,0.0031705615],"genre_scores_gemma":[0.9221166,0.00012971585,0.07385998,0.00011361561,0.000020770773,0.00016595326,0.00007657382,0.00008559732,0.0034311658],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99981207,0.000057193254,0.0000075591743,0.0000402401,0.000051460403,0.000031527394],"domain_scores_gemma":[0.99916875,0.000513254,0.00009821265,0.000047024852,0.000114984185,0.000057736626],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00063627947,0.00071846857,0.00066015485,0.00025345973,0.00025803593,0.0005799416,0.00073440414,0.0005750325,0.0021758026],"category_scores_gemma":[0.0022620384,0.00032524695,0.000345969,0.00022376065,0.0010277055,0.00056512054,0.0010576472,0.0009417618,0.00019360744],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000022431763,0.000012908347,0.00014880084,0.000028332131,0.0000098530045,0.000021868105,0.00003138103,0.9855258,0.0007954313,0.0073196082,0.00022966991,0.005853908],"study_design_scores_gemma":[0.0000051506345,0.0000122229085,0.000021993788,0.0000031860193,9.335417e-7,0.0000021893013,0.0000033017673,0.9972801,0.00012663048,0.0024200878,0.00012290991,0.0000012138046],"about_ca_topic_score_codex":0.0040538413,"about_ca_topic_score_gemma":0.0028688535,"teacher_disagreement_score":0.0040538413,"about_ca_system_score_codex":0.0008096093,"about_ca_system_score_gemma":0.0009347109,"threshold_uncertainty_score":0.008060455},"labels":[],"label_agreement":null},{"id":"W7132871321","doi":"","title":"Who Should I Trust? Uncertainty and Risk for Knowledge Transfer from Multiple Sources in Reinforcement Learning Domains","year":2023,"lang":"","type":"dissertation","venue":"TSpace","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Knowledge transfer; Transfer of learning; Set (abstract data type); Inference; Quality (philosophy); Bayesian inference; Uncertainty quantification; Bayesian probability","score_opus":0.042537842236209045,"score_gpt":0.326914110571627,"score_spread":0.28437626833541796,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7132871321","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06284183,0.0007128947,0.9251029,0.0047230027,0.000048678034,0.000087297594,0.00009997194,0.00019801338,0.006185311],"genre_scores_gemma":[0.95271254,0.00039201148,0.044404775,0.00039781275,0.000086604436,0.00015111617,0.000061085564,0.00006942224,0.0017246301],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98906916,0.0063505275,0.0005066006,0.0017437447,0.001676428,0.0006535025],"domain_scores_gemma":[0.9214678,0.06734466,0.0049127787,0.0031742584,0.0018352967,0.0012650717],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016205424,0.0010807053,0.001712286,0.0011304467,0.0012126393,0.004430302,0.0022810597,0.0036552711,0.002931563],"category_scores_gemma":[0.08060551,0.0010504343,0.0012608255,0.0007970618,0.0064053927,0.009707193,0.0053596976,0.0063667516,0.0003425133],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003763443,0.0001446915,0.0042519835,0.00025860697,0.00020968392,0.0005735308,0.0013904889,0.46744046,0.0010124887,0.46710467,0.0012922818,0.055944756],"study_design_scores_gemma":[0.000029194083,0.000058584694,0.0005774013,0.000055496876,0.000026024485,0.00009547609,0.00009988476,0.4735937,0.0005346221,0.52418274,0.00071128295,0.00003551512],"about_ca_topic_score_codex":0.002715666,"about_ca_topic_score_gemma":0.001276409,"teacher_disagreement_score":0.016205424,"about_ca_system_score_codex":0.003259302,"about_ca_system_score_gemma":0.0015986436,"threshold_uncertainty_score":0.08570355},"labels":[],"label_agreement":null},{"id":"W7132969921","doi":"","title":"Learning the Discount Factor in Inverse Reinforcement Learning with Application to Animal Behaviour","year":2023,"lang":"","type":"dissertation","venue":"TSpace","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Foraging; Reinforcement learning; Function (biology); Convexity; Preference; Factor (programming language)","score_opus":0.02195637573274086,"score_gpt":0.317475079676105,"score_spread":0.29551870394336416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7132969921","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02038585,0.00064759806,0.97528994,0.0004899694,0.000057324647,0.000040979685,0.00001783508,0.00009150664,0.0029791174],"genre_scores_gemma":[0.5692794,0.0015733723,0.4218743,0.00022230358,0.00014266872,0.00025446014,0.000057351386,0.0001504774,0.0064457646],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99939775,0.00028711767,0.000034588687,0.00013458551,0.000097501645,0.000048573987],"domain_scores_gemma":[0.99455476,0.0044524344,0.0003519181,0.0001825895,0.00030551173,0.00015279393],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028141018,0.0008934256,0.0010990519,0.00049763924,0.00044464227,0.0011655283,0.0011668691,0.00136743,0.0026147184],"category_scores_gemma":[0.013304031,0.00048509895,0.0008792648,0.0006121913,0.0022126073,0.0023104502,0.0014768365,0.0032367462,0.00023501771],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007175881,0.00009976111,0.0012538709,0.00018720824,0.000049989296,0.00008371049,0.00019577987,0.7717932,0.0015513427,0.1688669,0.0007658102,0.0550806],"study_design_scores_gemma":[0.000010423249,0.00002710284,0.000096682816,0.000013267879,0.000007415046,0.000011341824,0.000011325427,0.95928985,0.00025360376,0.03981378,0.00045568668,0.000009474135],"about_ca_topic_score_codex":0.0041057584,"about_ca_topic_score_gemma":0.0028167255,"teacher_disagreement_score":0.0041057584,"about_ca_system_score_codex":0.0017600867,"about_ca_system_score_gemma":0.001091321,"threshold_uncertainty_score":0.014882565},"labels":[],"label_agreement":null},{"id":"W7133015284","doi":"","title":"Acquiring Information from Bayesian Surprise in Cognitive Linear Gaussian Dynamic Systems","year":2023,"lang":"","type":"dissertation","venue":"TSpace","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Surprise; Novelty; Bayesian probability; Gaussian process; State (computer science); Gaussian; Cognition; Credibility","score_opus":0.023101257043908772,"score_gpt":0.33480495346296973,"score_spread":0.311703696419061,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7133015284","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.074138634,0.0002418161,0.92271554,0.00024724082,0.000019257292,0.00003915176,0.000029034778,0.00032143595,0.0022479193],"genre_scores_gemma":[0.9432541,0.00018650033,0.055340312,0.000086438886,0.00001953097,0.00006543225,0.00005175777,0.0000311588,0.0009648029],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9991628,0.00022622936,0.000045326626,0.00018698031,0.0002693731,0.00010916662],"domain_scores_gemma":[0.9959132,0.0029692955,0.00045085364,0.00016994223,0.0003524275,0.0001440997],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017247235,0.0007756116,0.0010539608,0.00050835346,0.00052023976,0.0014697417,0.0008965066,0.0009868254,0.00095513836],"category_scores_gemma":[0.00849842,0.00039346793,0.0006102438,0.0005083595,0.0012469455,0.0015009386,0.0016065347,0.0013535074,0.00014658188],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016002738,0.00004532289,0.0011569264,0.00008044557,0.000037179616,0.00010959278,0.00015889398,0.93690896,0.0024351885,0.022358026,0.0002822598,0.036267184],"study_design_scores_gemma":[0.000006476187,0.000030447458,0.00019106694,0.0000037217076,0.000007598809,0.000017476803,0.000009327755,0.99057055,0.0006565565,0.00839249,0.0001063749,0.000007988075],"about_ca_topic_score_codex":0.005753849,"about_ca_topic_score_gemma":0.0029013527,"teacher_disagreement_score":0.005753849,"about_ca_system_score_codex":0.0012831056,"about_ca_system_score_gemma":0.0014895152,"threshold_uncertainty_score":0.011440694},"labels":[],"label_agreement":null},{"id":"W7133070489","doi":"","title":"Optimizing Mechanism Design in Multi-Agent Reinforcement Learning","year":2024,"lang":"","type":"dissertation","venue":"TSpace","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Agencia Nacional de Investigación y Desarrollo; University of Toronto","keywords":"Reinforcement learning; Mechanism design; Markov decision process; Bridging (networking); Benchmark (surveying); Robustness (evolution); Mechanism (biology); Nash equilibrium","score_opus":0.06547687145312808,"score_gpt":0.3433784059699186,"score_spread":0.2779015345167905,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7133070489","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015300891,0.00052704645,0.9806113,0.0004968474,0.000043071588,0.00013153374,0.00003898924,0.00017835373,0.0026719403],"genre_scores_gemma":[0.8052276,0.00074889854,0.18941137,0.00028838337,0.000065535176,0.00079725543,0.000090429916,0.0000701728,0.0033003967],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99736744,0.001592396,0.00011960625,0.00036558838,0.00034312482,0.00021190092],"domain_scores_gemma":[0.992292,0.00606899,0.00061383157,0.0003147914,0.0004835382,0.00022689732],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005653471,0.0013214068,0.0017700588,0.00082090346,0.0005333758,0.0016534949,0.0018086099,0.0018306673,0.0021837635],"category_scores_gemma":[0.01294007,0.0007658019,0.00090865983,0.0007121358,0.0021372917,0.0018441429,0.0014546086,0.0023787562,0.00027990073],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000037032365,0.000049857113,0.00036700943,0.000088012646,0.00005054811,0.00003921283,0.000046834593,0.9297471,0.00033059638,0.059286956,0.0003401306,0.009616683],"study_design_scores_gemma":[0.000031529264,0.000028999866,0.00003330014,0.0000100786165,0.000008252579,0.0000065713093,0.0000061937894,0.971798,0.00012921868,0.027573477,0.0003693143,0.0000049673376],"about_ca_topic_score_codex":0.0026819676,"about_ca_topic_score_gemma":0.002064918,"teacher_disagreement_score":0.005653471,"about_ca_system_score_codex":0.0022983546,"about_ca_system_score_gemma":0.002324122,"threshold_uncertainty_score":0.029898763},"labels":[],"label_agreement":null},{"id":"W7133567929","doi":"10.1109/etfg61999.2025.11402119","title":"A Hybrid Learning Framework for State Estimation and Volt/Var Optimization Using Normalizing Flows and Reinforcement Learning","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Reinforcement learning; State (computer science); Stability (learning theory); Estimation; Noise (video); Key (lock); Control (management); Active learning (machine learning)","score_opus":0.01844932768772126,"score_gpt":0.2829148484919575,"score_spread":0.26446552080423624,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7133567929","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0017015118,0.00005754449,0.9974995,0.000031300126,0.000008713758,0.00001084573,0.00000750658,0.00011103135,0.0005720904],"genre_scores_gemma":[0.65300727,0.00033032493,0.34222785,0.00012272666,0.00011525651,0.00027748608,0.00011561759,0.0001158569,0.0036875436],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99942774,0.00018637891,0.000029543811,0.0001306172,0.00016858979,0.00005714074],"domain_scores_gemma":[0.99921477,0.00042437893,0.00009989182,0.00005632704,0.00016736824,0.000037247657],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014159145,0.0010028105,0.0010997568,0.00053349446,0.00034196582,0.0008455238,0.0014765589,0.00079637585,0.0015866961],"category_scores_gemma":[0.002080623,0.00039614332,0.000646945,0.000475106,0.0009658106,0.0009702334,0.0011605124,0.001243683,0.00029663846],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000022524051,0.000035472665,0.00028101506,0.00003238919,0.000027452304,0.00003299974,0.0000338685,0.93703383,0.0010103281,0.018227521,0.00034366225,0.042918872],"study_design_scores_gemma":[0.0000024984704,0.000011122462,0.000024278937,0.0000021422868,0.0000023090727,0.0000042094734,0.0000013306799,0.9972324,0.00014293563,0.0023904124,0.000183832,0.000002446377],"about_ca_topic_score_codex":0.007136042,"about_ca_topic_score_gemma":0.005136121,"teacher_disagreement_score":0.007136042,"about_ca_system_score_codex":0.0008238421,"about_ca_system_score_gemma":0.0011864632,"threshold_uncertainty_score":0.014189005},"labels":[],"label_agreement":null},{"id":"W7134200277","doi":"10.1109/bigdata66926.2025.11402252","title":"Goal-Conditioned Reinforcement Learning for Data-Driven Maritime Navigation","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Reinforcement learning; Control (management); Action (physics); Field (mathematics); Stability (learning theory)","score_opus":0.028944629782112315,"score_gpt":0.3041871407674779,"score_spread":0.27524251098536556,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7134200277","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032251444,0.00019217869,0.9652154,0.00016536105,0.00006636364,0.000039404178,0.00006696967,0.0006190053,0.0013838008],"genre_scores_gemma":[0.9510553,0.00007343607,0.04707665,0.000072915194,0.000019844358,0.00008393288,0.000089839596,0.000057905207,0.0014702652],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99967325,0.00009933324,0.00001579542,0.000071096365,0.00008405961,0.000056497203],"domain_scores_gemma":[0.998844,0.0006428594,0.00009499346,0.00009066614,0.00023088422,0.0000965332],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010026753,0.00053876766,0.0008379834,0.00023144444,0.00027603077,0.0004924994,0.0011359893,0.000679849,0.0019805508],"category_scores_gemma":[0.0034979577,0.00033748665,0.00034204568,0.0002564354,0.00079911476,0.0006343228,0.001247021,0.0015025281,0.00026028568],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016879641,0.00008533441,0.00049876404,0.000050941995,0.00002839722,0.00003521683,0.000032631055,0.95238477,0.0021933839,0.0075772665,0.0009220648,0.036022462],"study_design_scores_gemma":[0.0000072010557,0.00001648841,0.00003666237,0.0000016515124,0.0000018551302,0.0000021521346,0.0000010911385,0.9970546,0.00025158696,0.0025510343,0.00007380198,0.0000018806352],"about_ca_topic_score_codex":0.006798736,"about_ca_topic_score_gemma":0.0060748355,"teacher_disagreement_score":0.006798736,"about_ca_system_score_codex":0.0007160925,"about_ca_system_score_gemma":0.0012170938,"threshold_uncertainty_score":0.013518333},"labels":[],"label_agreement":null},{"id":"W7134938218","doi":"10.1109/aiot66900.2025.00136","title":"A Comparative Evaluation of Teacher-Guided Reinforcement Learning Techniques for Autonomous Cyber Operations","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Royal Military College of Canada","funders":"","keywords":"Reinforcement learning; Automation; Control (management); Key (lock); Action (physics)","score_opus":0.08454914817489433,"score_gpt":0.38284402444833376,"score_spread":0.2982948762734394,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7134938218","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7101893,0.0019504572,0.27084887,0.0003561159,0.00012499811,0.0003208964,0.00012328138,0.0028477726,0.013238392],"genre_scores_gemma":[0.95654863,0.00031739747,0.041122742,0.00002785709,0.000012477486,0.00006395333,0.000075501535,0.000082693754,0.0017487722],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99918264,0.00038162494,0.00004193527,0.00009828551,0.00023672591,0.000058819063],"domain_scores_gemma":[0.99459535,0.0039578606,0.00019445983,0.00038128367,0.0006995329,0.00017151546],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015916001,0.0005433168,0.00065183424,0.0005031832,0.00029373777,0.00047770128,0.0010410274,0.0008410252,0.0019909996],"category_scores_gemma":[0.006233651,0.0002022723,0.00025097182,0.00033225247,0.0004520021,0.0008293134,0.0006235719,0.00079884066,0.00030009376],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024902395,0.0018165535,0.0021866467,0.00061275245,0.00014210358,0.000059189646,0.00022059899,0.5375456,0.009109633,0.0021597499,0.0010872474,0.44256973],"study_design_scores_gemma":[0.00017960276,0.001681273,0.0013106723,0.000018202192,0.000039471484,0.00004158687,0.00005670396,0.98904234,0.0058804937,0.0007542211,0.0009804231,0.000015033812],"about_ca_topic_score_codex":0.0048439982,"about_ca_topic_score_gemma":0.004153318,"teacher_disagreement_score":0.0048439982,"about_ca_system_score_codex":0.0007221706,"about_ca_system_score_gemma":0.0008420929,"threshold_uncertainty_score":0.009631634},"labels":[],"label_agreement":null},{"id":"W7136135934","doi":"10.1109/itsc60802.2025.11423815","title":"Optimizing Pedestrian Safety in Real-Time: An Extreme Value Theory-Based Reinforcement Learning Framework","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Pedestrian; Reinforcement learning; Value (mathematics); Extreme learning machine; Control (management); Event (particle physics)","score_opus":0.026060300312092624,"score_gpt":0.2863342966971705,"score_spread":0.26027399638507787,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7136135934","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025401924,0.0003963277,0.9696363,0.00035762187,0.00005779904,0.000050118935,0.000033416254,0.00019438335,0.0038720304],"genre_scores_gemma":[0.95115966,0.0002651247,0.04474408,0.00018393167,0.00006370935,0.00016829318,0.000061043145,0.000035913006,0.0033182632],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992575,0.00031360087,0.000029066652,0.00013701647,0.00014588179,0.000116822004],"domain_scores_gemma":[0.9984036,0.0009801014,0.00019927973,0.0000376193,0.0002656031,0.00011382883],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016380931,0.0011654678,0.0014303263,0.00049607706,0.0003204012,0.0010787335,0.0016523171,0.0014176794,0.0019108297],"category_scores_gemma":[0.0030588883,0.0005044261,0.0008093659,0.00037826176,0.0012625139,0.00068782753,0.0011689005,0.001583611,0.00025876518],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000020037978,0.000025457617,0.00034763385,0.000022130938,0.000019589112,0.000045199886,0.00001994882,0.99102026,0.00021218193,0.003319845,0.00016817354,0.004779674],"study_design_scores_gemma":[0.000005141407,0.000017808152,0.00004031411,0.0000029037121,0.0000039705697,0.0000039016636,0.0000027555027,0.99841607,0.000038958802,0.0013899576,0.000075675154,0.0000024900517],"about_ca_topic_score_codex":0.008542538,"about_ca_topic_score_gemma":0.0041097733,"teacher_disagreement_score":0.008542538,"about_ca_system_score_codex":0.0011116079,"about_ca_system_score_gemma":0.0013141738,"threshold_uncertainty_score":0.016985655},"labels":[],"label_agreement":null},{"id":"W7139921426","doi":"10.1109/acoit66109.2025.11437087","title":"Emergent AI Behaviors in Multi-Modal Fusion Models: Risks, Patterns, and Control","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Oakville-Trafalgar Memorial Hospital","funders":"","keywords":"Control (management); Sensor fusion; Fusion; Key (lock); Control system","score_opus":0.04504473877454069,"score_gpt":0.3236207202977955,"score_spread":0.2785759815232548,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7139921426","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09895133,0.0003207924,0.8848908,0.0023167394,0.00005857645,0.0000844681,0.00007783139,0.00023306582,0.01306631],"genre_scores_gemma":[0.9472213,0.00017899112,0.04959631,0.00013855178,0.000028374521,0.00013177641,0.00004784871,0.00004753914,0.0026092613],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99798006,0.00088288926,0.00009776621,0.00033036628,0.00048696483,0.00022194277],"domain_scores_gemma":[0.9921984,0.004480634,0.0012040072,0.0012146811,0.00049466436,0.0004076092],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038452516,0.0006663056,0.00067449064,0.0005757235,0.0010323816,0.002489488,0.0013185159,0.0014281472,0.0022475328],"category_scores_gemma":[0.0126754185,0.00044611312,0.0009918266,0.00037859008,0.0051743933,0.0041051563,0.0042186775,0.002815723,0.00020142374],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000095040064,0.00006224954,0.0031920776,0.00009901962,0.00008055691,0.0003555025,0.0013762283,0.3401559,0.002667822,0.63844144,0.00065437076,0.012819771],"study_design_scores_gemma":[0.000010993568,0.00003692574,0.000315964,0.000022426813,0.00001433807,0.000065003274,0.00019125798,0.58308595,0.0005115352,0.41485876,0.00086551846,0.00002129363],"about_ca_topic_score_codex":0.0028305897,"about_ca_topic_score_gemma":0.002070459,"teacher_disagreement_score":0.0038452516,"about_ca_system_score_codex":0.0016410383,"about_ca_system_score_gemma":0.0011688349,"threshold_uncertainty_score":0.020335853},"labels":[],"label_agreement":null},{"id":"W7143646775","doi":"10.71465/ajainn17","title":"Advances in Reinforcement Learning for Autonomous Systems","year":2020,"lang":"","type":"article","venue":"American Journal of Artificial Intelligence and Neural Networks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Reinforcement learning; Autonomous system (mathematics); Reinforcement; Autonomous learning; Autonomous agent","score_opus":0.040690444154212566,"score_gpt":0.29102130276723503,"score_spread":0.25033085861302246,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7143646775","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0036002707,0.05252477,0.9050002,0.006474872,0.0012623738,0.000050292165,0.000090174,0.00032878722,0.030668277],"genre_scores_gemma":[0.5543076,0.12048061,0.2922224,0.002224885,0.0043716324,0.00044842614,0.00037483702,0.00023202112,0.025337638],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9992931,0.00024758553,0.000051584106,0.00009897534,0.00026661943,0.00004208751],"domain_scores_gemma":[0.998541,0.0009867213,0.00007622812,0.00010672733,0.00023517494,0.000054185493],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013999309,0.00094372575,0.0009737325,0.00045445366,0.0003046159,0.0015637267,0.00095746137,0.0012749194,0.0031198808],"category_scores_gemma":[0.0038016343,0.00031513724,0.000703322,0.00071311137,0.0013994391,0.0019301176,0.0012369509,0.003580568,0.0009198385],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000053776643,0.000064347485,0.00059929595,0.0006337202,0.00009741965,0.00012833813,0.00015481818,0.21027067,0.001310934,0.5906559,0.010181178,0.18584949],"study_design_scores_gemma":[0.000039883253,0.00007884664,0.00026329028,0.00020663899,0.000031278643,0.00008864484,0.000039682553,0.48409483,0.00074839435,0.4430937,0.071283065,0.000031802432],"about_ca_topic_score_codex":0.0023242745,"about_ca_topic_score_gemma":0.0013458623,"teacher_disagreement_score":0.0031198808,"about_ca_system_score_codex":0.0012472407,"about_ca_system_score_gemma":0.001174415,"threshold_uncertainty_score":0.010437071},"labels":[],"label_agreement":null},{"id":"W75018201","doi":"10.1007/978-0-387-74483-4_7","title":"GA Based Task Allocation Models","year":2008,"lang":"en","type":"book-chapter","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"St. Francis Xavier University","funders":"","keywords":"Task (project management); Computer science; Engineering; Systems engineering","score_opus":0.037697866988620175,"score_gpt":0.22151136079020498,"score_spread":0.1838134938015848,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W75018201","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009571129,0.0013787835,0.9544278,0.00046973937,0.00022857652,0.00006276995,0.00015473375,0.00075315,0.03295334],"genre_scores_gemma":[0.62597454,0.0030336962,0.2644528,0.00039184603,0.00014043087,0.0005946133,0.00050651445,0.00034639324,0.10455912],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99984527,0.000043978187,0.0000050587187,0.00003458867,0.000044125307,0.000026998532],"domain_scores_gemma":[0.99980813,0.000096974414,0.00001565349,0.000021496511,0.000042852273,0.000014863596],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00028606696,0.0007943094,0.00070799614,0.00030932148,0.0002648957,0.0009284667,0.0013536215,0.0009033788,0.007267885],"category_scores_gemma":[0.0008526678,0.00035502072,0.0004899524,0.0005053229,0.0004844162,0.0008059089,0.0005286676,0.0012155783,0.0016324734],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000027184304,0.00002953593,0.0000797207,0.00004698245,0.000021248608,0.000025515867,0.000019853991,0.9025292,0.00096993166,0.03671434,0.0039011475,0.055635385],"study_design_scores_gemma":[0.000005235237,0.000014944228,0.000040108636,0.000009308552,0.000005996232,0.000012548121,0.0000049225264,0.97768563,0.00028444038,0.019181473,0.0027501378,0.0000052836285],"about_ca_topic_score_codex":0.0038496214,"about_ca_topic_score_gemma":0.0044210805,"teacher_disagreement_score":0.007267885,"about_ca_system_score_codex":0.00076710817,"about_ca_system_score_gemma":0.00090257433,"threshold_uncertainty_score":0.02431351},"labels":[],"label_agreement":null},{"id":"W75140964","doi":"","title":"Agnostic KWIK learning and efficient approximate reinforcement learning","year":2011,"lang":"en","type":"article","venue":"Conference on Learning Theory","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Simple (philosophy); Artificial intelligence; Algorithm; Theoretical computer science; Machine learning","score_opus":0.034468881668759915,"score_gpt":0.247999907469349,"score_spread":0.21353102580058908,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W75140964","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0101804575,0.00015103583,0.98761785,0.00015356098,0.000023355427,0.00002352155,0.000015204711,0.00026823685,0.001566832],"genre_scores_gemma":[0.80927944,0.0002704483,0.18276688,0.00023236367,0.000058299425,0.0001744431,0.00012535797,0.00013030642,0.006962366],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99858475,0.00057231425,0.00007579694,0.00022445022,0.00038042196,0.00016220241],"domain_scores_gemma":[0.99745303,0.0013625252,0.00026759438,0.00049686804,0.0003151932,0.00010477109],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001860294,0.00087680673,0.001381903,0.0004678743,0.00037415893,0.0010179984,0.0019330666,0.001330218,0.0026726017],"category_scores_gemma":[0.009071812,0.000592013,0.00044640983,0.0005466783,0.0015692769,0.002378393,0.0020600322,0.002584088,0.00061499455],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001534725,0.00012702198,0.0007214198,0.00011032251,0.000056125886,0.00008379658,0.00009197497,0.8176853,0.0017360165,0.12523326,0.0014868611,0.052514415],"study_design_scores_gemma":[0.000010870319,0.000019229541,0.000034012974,0.000003815441,0.000003934866,0.000014025362,0.0000036554213,0.9684824,0.00029437363,0.030806553,0.00032332746,0.0000038496946],"about_ca_topic_score_codex":0.0023732483,"about_ca_topic_score_gemma":0.0022857839,"teacher_disagreement_score":0.0026726017,"about_ca_system_score_codex":0.0010398349,"about_ca_system_score_gemma":0.0011683039,"threshold_uncertainty_score":0.009838283},"labels":[],"label_agreement":null},{"id":"W75509695","doi":"","title":"A new Q(lambda) with interim forward view and Monte Carlo equivalence","year":2014,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Monte Carlo method; Markov decision process; Algorithm; Equivalence (formal languages); Q-learning; Markov process; Mathematical optimization; Artificial intelligence; Mathematics; Discrete mathematics; Statistics","score_opus":0.03106032801753098,"score_gpt":0.2875444306336761,"score_spread":0.25648410261614507,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W75509695","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0017217663,0.000105067294,0.9925323,0.00030779967,0.00010194256,0.000037123955,0.000055661778,0.0001096484,0.0050287833],"genre_scores_gemma":[0.15022542,0.00048080378,0.8360615,0.0009286797,0.00041127505,0.00033647323,0.00020158812,0.0004968784,0.0108574135],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997184,0.00062437734,0.00019798901,0.00067891506,0.0010650726,0.0002495894],"domain_scores_gemma":[0.9951552,0.0020657014,0.00031503625,0.00070684694,0.0013432403,0.0004139411],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040338608,0.0009085628,0.0010278344,0.0012357362,0.00090972317,0.0023494186,0.0031877295,0.0022551548,0.008540301],"category_scores_gemma":[0.015651155,0.0007992766,0.0016591643,0.0011781803,0.0035825898,0.0051007764,0.00490481,0.006115992,0.0016224579],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000028831319,0.000065859036,0.0005647421,0.00010088333,0.000029340918,0.000106835476,0.00015462589,0.054196812,0.001751577,0.89922476,0.0033794083,0.040396303],"study_design_scores_gemma":[0.000027206517,0.00007208112,0.00019143755,0.000044793807,0.000016476019,0.0001402528,0.000019921348,0.5310973,0.0011676246,0.45638606,0.010801957,0.000034920504],"about_ca_topic_score_codex":0.0028633033,"about_ca_topic_score_gemma":0.0024439434,"teacher_disagreement_score":0.008540301,"about_ca_system_score_codex":0.0022420827,"about_ca_system_score_gemma":0.004008454,"threshold_uncertainty_score":0.028570175},"labels":[],"label_agreement":null},{"id":"W80327629","doi":"10.2316/journal.206.2009.3.206-3270","title":"INTERNAL REPRESENTATION OF THE ENVIRONMENT IN COGNITIVE ROBOTICS","year":2009,"lang":"en","type":"article","venue":"International Journal of Robotics and Automation","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Representation (politics); Artificial intelligence; Cognitive robotics; Computer science; Cognition; Robotics; Psychology; Robot; Neuroscience; Political science","score_opus":0.017317283841722236,"score_gpt":0.2809598773564247,"score_spread":0.26364259351470243,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W80327629","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07070585,0.010313118,0.8562634,0.0058782576,0.00029817712,0.00004061961,0.00011463904,0.00032832063,0.056057636],"genre_scores_gemma":[0.88491666,0.0041006077,0.105705276,0.00046326878,0.00023816626,0.00012167277,0.00011560257,0.000051327104,0.0042873994],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995363,0.00016427014,0.00002155748,0.000100141,0.00012958627,0.000048087506],"domain_scores_gemma":[0.99949014,0.00023329035,0.00005647425,0.00011683483,0.000059725655,0.00004352927],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007951167,0.00045980708,0.0004987823,0.0008068029,0.00051746285,0.0026482807,0.0009916173,0.001369364,0.0013327503],"category_scores_gemma":[0.002347907,0.00030589555,0.00055775966,0.0006905214,0.0063176295,0.005160221,0.0016453845,0.0015469114,0.00026737546],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001140935,0.000009792421,0.00022905586,0.000084521685,0.000017109407,0.000056412795,0.00053573167,0.014429768,0.00093524193,0.9657138,0.0003316114,0.017645575],"study_design_scores_gemma":[0.000006951826,0.000014020232,0.0002958386,0.000025730431,0.00000834231,0.000045165547,0.00009381646,0.022966405,0.0003353745,0.97264975,0.0035412614,0.000017302953],"about_ca_topic_score_codex":0.0014687923,"about_ca_topic_score_gemma":0.00081396627,"teacher_disagreement_score":0.0026482807,"about_ca_system_score_codex":0.00091343315,"about_ca_system_score_gemma":0.0006474174,"threshold_uncertainty_score":0.006627381},"labels":[],"label_agreement":null},{"id":"W84951255","doi":"","title":"Learning subjective representations for planning","year":2005,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; University of Waterloo","funders":"","keywords":"Embedding; Representation (politics); Computer science; Artificial intelligence; Domain (mathematical analysis); Construct (python library); Action (physics); Domain knowledge; Robot; Sequence (biology); Knowledge representation and reasoning; Machine learning; Mathematics","score_opus":0.02995310586468893,"score_gpt":0.31831261305652775,"score_spread":0.28835950719183884,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W84951255","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0059863473,0.00009810751,0.9927176,0.00010159799,0.000012550026,0.000016501577,0.00005142013,0.00017555764,0.0008402588],"genre_scores_gemma":[0.56841046,0.00059321953,0.42760652,0.0001469797,0.000069636255,0.00018224308,0.000595633,0.000117594085,0.0022776532],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99938023,0.00030031436,0.000039424503,0.00012765404,0.00011810946,0.000034309945],"domain_scores_gemma":[0.99782866,0.0012830731,0.00023372716,0.00036276158,0.00019267593,0.00009900104],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010806991,0.00076300977,0.00045496796,0.0005104824,0.00022293144,0.0009519854,0.00083142484,0.000639896,0.001843218],"category_scores_gemma":[0.0068023237,0.00036111742,0.00052529114,0.0005184502,0.001163803,0.003066364,0.0012020905,0.0015741326,0.00034954087],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014949012,0.00009426668,0.0009890604,0.00022918124,0.000070999915,0.00009229586,0.00028171405,0.5139337,0.004736778,0.2673538,0.0024568446,0.20961176],"study_design_scores_gemma":[0.000009918505,0.000047945727,0.00013220424,0.0000154209,0.000008668427,0.000019952358,0.0000199781,0.88402784,0.0011639196,0.11332164,0.0012213178,0.000011288694],"about_ca_topic_score_codex":0.0009654918,"about_ca_topic_score_gemma":0.0015535717,"teacher_disagreement_score":0.001843218,"about_ca_system_score_codex":0.0006872182,"about_ca_system_score_gemma":0.00049381226,"threshold_uncertainty_score":0.00616616},"labels":[],"label_agreement":null},{"id":"W85998123","doi":"10.1609/icaps.v19i1.13355","title":"Incremental Policy Generation for Finite-Horizon DEC-POMDPs","year":2009,"lang":"en","type":"article","venue":"Proceedings of the International Conference on Automated Planning and Scheduling","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":64,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"National Science Foundation","keywords":"Dynamic programming; Computer science; Scalability; Reachability; Mathematical optimization; State (computer science); State space; Reduction (mathematics); Backup; Horizon; Algorithm; Mathematics","score_opus":0.05613533112334739,"score_gpt":0.31798258634313714,"score_spread":0.26184725521978974,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W85998123","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039699115,0.00027396553,0.9554786,0.00021913878,0.00005236039,0.000090081325,0.00013830481,0.0012788428,0.0027695992],"genre_scores_gemma":[0.7471653,0.00020560618,0.25092876,0.00011312789,0.000017575823,0.00021881527,0.00030885285,0.000114270544,0.0009276991],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995235,0.00016684677,0.000028962284,0.0000823141,0.0001228299,0.000075575794],"domain_scores_gemma":[0.9972832,0.0020494794,0.00015516367,0.00022697092,0.00018883172,0.00009638236],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015012231,0.0005849361,0.0008204827,0.00044208847,0.00047025792,0.00070936105,0.001140796,0.00064729765,0.0018524624],"category_scores_gemma":[0.0047778855,0.00046045138,0.00046595422,0.00043166298,0.00067823764,0.0011990326,0.0011052081,0.0013301382,0.00024064542],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000119383105,0.00005961927,0.0005233021,0.00008821355,0.000020911017,0.00008614371,0.000049023234,0.9473028,0.0009667077,0.01009326,0.000782795,0.0399078],"study_design_scores_gemma":[0.0000108454005,0.000015116728,0.000035130925,0.000004241344,0.000002989667,0.000011822188,0.000008866974,0.9933541,0.00055450184,0.005666673,0.0003329701,0.0000026954533],"about_ca_topic_score_codex":0.0027833725,"about_ca_topic_score_gemma":0.0035378158,"teacher_disagreement_score":0.0027833725,"about_ca_system_score_codex":0.0008007422,"about_ca_system_score_gemma":0.0012970422,"threshold_uncertainty_score":0.007939279},"labels":[],"label_agreement":null},{"id":"W87408738","doi":"","title":"Apprentissage de la coordination multiagent : Q-learning par jeu adaptatif","year":2005,"lang":"fr","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Humanities; Nash equilibrium; Mathematical economics; Philosophy; Mathematics","score_opus":0.02170017316396445,"score_gpt":0.274899334009066,"score_spread":0.2531991608451016,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W87408738","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.30040866,0.0006770656,0.684841,0.0010226335,0.00023874907,0.0003406112,0.000052793817,0.0009450209,0.0114735095],"genre_scores_gemma":[0.9079118,0.00017326912,0.08539956,0.00014998754,0.000044353194,0.00018810086,0.00004141891,0.000052571253,0.006038958],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986743,0.0005020684,0.000054576525,0.00026833705,0.00034347156,0.00015730434],"domain_scores_gemma":[0.99576366,0.0025395795,0.00026830443,0.00039466962,0.0007710909,0.00026271353],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030120984,0.00084972446,0.0009853786,0.00064879697,0.0011208609,0.00167333,0.0015800227,0.001663674,0.0029335693],"category_scores_gemma":[0.008423,0.00038527735,0.0006217459,0.0005391288,0.0015557677,0.0012740969,0.0015112781,0.0018783592,0.0005095704],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003710505,0.00042352412,0.0047504045,0.00015089981,0.00014524146,0.00035982917,0.0007544731,0.8543552,0.0054032602,0.027020134,0.0017745649,0.10449134],"study_design_scores_gemma":[0.00004153666,0.00010629198,0.000543067,0.000010398347,0.000015886855,0.000049343318,0.000053993186,0.9930206,0.0012384207,0.0038809713,0.001026416,0.000013055066],"about_ca_topic_score_codex":0.010489368,"about_ca_topic_score_gemma":0.0045079286,"teacher_disagreement_score":0.010489368,"about_ca_system_score_codex":0.0010745033,"about_ca_system_score_gemma":0.00149231,"threshold_uncertainty_score":0.020856619},"labels":[],"label_agreement":null},{"id":"W88806592","doi":"10.1007/978-3-642-28499-1_5","title":"Leveraging Domain Knowledge to Learn Normative Behavior: A Bayesian Approach","year":2012,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick; University of Waterloo","funders":"","keywords":"Normative; Computer science; Reinforcement learning; Norm (philosophy); Domain knowledge; Artificial intelligence; Set (abstract data type); Domain (mathematical analysis); Adaptation (eye); Process (computing); Normative model of decision-making; Bayesian probability; Machine learning; Knowledge management; Psychology; Epistemology; Mathematics","score_opus":0.02530434686639648,"score_gpt":0.26163652532155574,"score_spread":0.23633217845515925,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W88806592","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008152233,0.00021946536,0.98823684,0.00047777456,0.00002158362,0.00003812681,0.00007113808,0.00017094343,0.0026118376],"genre_scores_gemma":[0.59573275,0.001035236,0.39698386,0.00044219234,0.0002064226,0.0004325044,0.0004242798,0.00019202044,0.004550749],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99769515,0.0010184954,0.0001348314,0.0005129218,0.0004966326,0.0001419285],"domain_scores_gemma":[0.9871842,0.010025947,0.0007695771,0.00096898293,0.00077589776,0.00027544625],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0050423155,0.001378257,0.0020608264,0.0017876083,0.0005678391,0.0023019088,0.0031640504,0.0021622097,0.0033424355],"category_scores_gemma":[0.020728063,0.0012877298,0.0014793518,0.0012811738,0.0024072325,0.005554229,0.0023268182,0.004565438,0.0006022935],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016829619,0.00035695094,0.002831228,0.00029769226,0.00037830762,0.00012241471,0.00033927482,0.572278,0.0019946327,0.23720986,0.0031881148,0.18083528],"study_design_scores_gemma":[0.000012915407,0.000029529623,0.00026240153,0.000034943972,0.00002643393,0.00002312534,0.000017453325,0.7740043,0.00038246126,0.22471778,0.0004685504,0.000020146595],"about_ca_topic_score_codex":0.0030014892,"about_ca_topic_score_gemma":0.0049427557,"teacher_disagreement_score":0.0050423155,"about_ca_system_score_codex":0.0013476921,"about_ca_system_score_gemma":0.0015774792,"threshold_uncertainty_score":0.026666641},"labels":[],"label_agreement":null},{"id":"W9932698","doi":"10.48550/arxiv.1205.2619","title":"Regret-based Reward Elicitation for Markov Decision Processes","year":2012,"lang":"en","type":"article","venue":"Uncertainty in Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":61,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Regret; Markov decision process; Computer science; Preference elicitation; Function (biology); Minimax; Preference; Markov process; Markov chain; Mathematical optimization; Machine learning; Mathematics; Economics; Microeconomics","score_opus":0.06076316977610513,"score_gpt":0.3316740532104062,"score_spread":0.2709108834343011,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W9932698","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018666958,0.00031697107,0.9776177,0.00044228762,0.00001983315,0.00012885022,0.00011256472,0.00027109,0.002423646],"genre_scores_gemma":[0.72220886,0.0005815595,0.27363673,0.0003022779,0.00006794183,0.00059048145,0.00035680362,0.00013881378,0.0021166543],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9872334,0.008725257,0.00060061825,0.0009218833,0.0019952029,0.00052365195],"domain_scores_gemma":[0.96617883,0.02893035,0.0018995128,0.0013410148,0.0009995678,0.0006508022],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011433134,0.0015079252,0.0015770271,0.00082163914,0.0006791665,0.002103598,0.0011756781,0.0017245708,0.0027358113],"category_scores_gemma":[0.041986015,0.0009217853,0.0012498028,0.0009218327,0.0017412058,0.0026677358,0.0025535296,0.003180852,0.00040626316],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003249433,0.0001318848,0.00071712554,0.00020349641,0.000063731844,0.00010459884,0.00022319166,0.87428635,0.0016521846,0.09275765,0.0008459125,0.028688956],"study_design_scores_gemma":[0.00003686524,0.000060655544,0.0001073743,0.000023787328,0.000009743308,0.000018610837,0.000015950924,0.92830133,0.00085179973,0.07011199,0.00044815795,0.000013751724],"about_ca_topic_score_codex":0.002232675,"about_ca_topic_score_gemma":0.0023373375,"teacher_disagreement_score":0.011433134,"about_ca_system_score_codex":0.0031546152,"about_ca_system_score_gemma":0.002566503,"threshold_uncertainty_score":0.06046492},"labels":[],"label_agreement":null}]}