{"id":"W3208994216","doi":"10.48550/arxiv.2110.14096","title":"Towards Robust Bisimulation Metric Learning","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Embedding; Computer science; Robustness (evolution); Artificial intelligence; Representation (politics); Theoretical computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0003895854,0.0003382746,0.0003682099,0.0006468992,0.0002364657,0.0004820013,0.00176538,0.0003611321,0.00007916758],"category_scores_gemma":[0.0002654251,0.0004240672,0.0002626849,0.00191919,0.00005473252,0.0006300226,0.003281706,0.001195026,0.00009245821],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003753191,"about_ca_system_score_gemma":0.0003323302,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000140098,"about_ca_topic_score_gemma":0.000006147483,"domain_scores_codex":[0.9976774,0.0002759965,0.0002699535,0.001140115,0.0002262504,0.0004103186],"domain_scores_gemma":[0.9977611,0.0001525909,0.0003978298,0.001189135,0.0003400145,0.0001593214],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000003775313,0.00002060383,0.003206037,0.00005268929,0.00007514169,0.000194044,0.0002030902,0.9801795,0.000005894614,0.01509755,0.00003455885,0.0009270458],"study_design_scores_gemma":[0.000279557,0.00004690257,0.002286286,0.00007912477,0.00005850445,0.000003169121,0.00009061003,0.9955764,0.00005506533,0.0005674161,0.0005241384,0.0004328216],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04006473,0.00008208097,0.9525794,0.00006009148,0.000855175,0.0001858386,5.973545e-7,0.000453042,0.005719096],"genre_scores_gemma":[0.9818867,0.0001799851,0.01367927,0.00006060403,0.00007754454,4.556529e-7,0.00004173125,0.00002472915,0.00404901],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9418219,"threshold_uncertainty_score":0.9998211,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09633591901691252,"score_gpt":0.1997533717924085,"score_spread":0.103417452775496,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}