{"id":"W4401414318","doi":"10.1109/icra57147.2024.10610065","title":"ISR-LLM: Iterative Self-Refined Large Language Model for Long-Horizon Sequential Task Planning","year":2024,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"JST-Mirai Program; Natural Sciences and Engineering Research Council of Canada","keywords":"Task (project management); Computer science; Horizon; Time horizon; Artificial intelligence; Mathematical optimization; Engineering; Mathematics; Systems engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00217705,0.00122293,0.00088273,0.0008248173,0.0005868163,0.001185545,0.002957492,0.001070879,0.004998809],"category_scores_gemma":[0.006722506,0.0008028322,0.00194953,0.0006224334,0.001259478,0.00240451,0.00263126,0.002336698,0.001502866],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001499862,"about_ca_system_score_gemma":0.004550663,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01355092,"about_ca_topic_score_gemma":0.02352795,"domain_scores_codex":[0.9984506,0.0006030898,0.0001243334,0.0002827792,0.0004220142,0.0001170059],"domain_scores_gemma":[0.9973067,0.00157354,0.0002033297,0.0005144763,0.0002961323,0.000105875],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003306611,0.0002029281,0.000907694,0.0004996454,0.00008573397,0.0002893113,0.0007927191,0.7695464,0.01031506,0.03671846,0.009866292,0.1704452],"study_design_scores_gemma":[0.000032728,0.00004154113,0.00005371973,0.00001379113,0.00001287775,0.00002362222,0.00002706295,0.9842868,0.002298004,0.009808131,0.003388991,0.00001277179],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004916817,0.0001121874,0.9843592,0.0001704113,0.0000284805,0.0001656424,0.0002652012,0.008797391,0.001184647],"genre_scores_gemma":[0.1618402,0.000148247,0.8330274,0.0001934041,0.00002387678,0.0006140036,0.001266034,0.001021413,0.001865364],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01355092,"threshold_uncertainty_score":0.02694404,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02641663869101564,"score_gpt":0.3070668637932958,"score_spread":0.2806502251022802,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}