{"id":"W7134200277","doi":"10.1109/bigdata66926.2025.11402252","title":"Goal-Conditioned Reinforcement Learning for Data-Driven Maritime Navigation","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Reinforcement learning; Control (management); Action (physics); Field (mathematics); Stability (learning theory)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.001306688,0.000530225,0.0005463795,0.0004111976,0.001283403,0.00137712,0.003222867,0.0003061469,0.0005858872],"category_scores_gemma":[0.000567754,0.0005958797,0.0002008887,0.001050407,0.0001879944,0.00241424,0.002802199,0.0006771632,0.0003960776],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003770512,"about_ca_system_score_gemma":0.000630951,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00008765583,"about_ca_topic_score_gemma":0.000005259773,"domain_scores_codex":[0.9951238,0.0002086768,0.001427185,0.001453845,0.0007917776,0.0009947278],"domain_scores_gemma":[0.9954401,0.0006644838,0.0006213898,0.002504224,0.0005820577,0.0001878025],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003713528,0.00003988911,0.0002884727,0.0003073983,0.0002272228,0.000004242896,0.0002982173,0.8605395,0.0001515787,0.1104564,0.01216272,0.01548726],"study_design_scores_gemma":[0.00174545,0.0004100876,0.0002526265,0.000517438,0.0001664581,0.000004640321,0.000110958,0.9281956,0.0004433985,0.0008139504,0.06679982,0.0005395962],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0001191002,0.0001027375,0.9630099,0.002564705,0.002107464,0.001877134,0.000008988767,0.0003884871,0.02982146],"genre_scores_gemma":[0.7381689,0.0001246773,0.09342303,0.001524135,0.0002688279,0.0001629917,0.003329695,0.00004863565,0.1629491],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8695869,"threshold_uncertainty_score":0.9996595,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02894462978211232,"score_gpt":0.3041871407674779,"score_spread":0.2752425109853656,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}